From 1d5da380fc2f3177e47d5a83e21f0ea0b0bea578 Mon Sep 17 00:00:00 2001 From: Birdmachine Date: Fri, 17 Jul 2026 09:21:09 -0400 Subject: [PATCH 1/3] switch assay metadata pages to Harmonized set (& purge holding Testing directory) --- docs/assays/{metadata/testing => }/.directory | 2 +- .../metadata/{testing => }/10X-Multiome.md | 0 docs/assays/metadata/10XMultiome.md | 24 - docs/assays/metadata/4i.md | 48 +- docs/assays/metadata/ATACseq.md | 351 ++++-------- .../{testing => }/Auto-fluorescence.md | 0 docs/assays/metadata/AutoFluorescence.md | 105 ---- docs/assays/metadata/CODEX.md | 182 +++--- docs/assays/metadata/COMET.md | 37 -- .../metadata/{testing => }/Cell-DIVE.md | 0 docs/assays/metadata/CosMx-Proteomics.md | 43 -- docs/assays/metadata/CosMx-Transcriptomics.md | 88 --- docs/assays/metadata/CyCIF.md | 34 -- docs/assays/metadata/CyTOF.md | 38 -- docs/assays/metadata/DESI.md | 112 ++-- docs/assays/metadata/DNA-Methylation.md | 28 - docs/assays/metadata/EnhancedSRS.md | 34 -- docs/assays/metadata/FACS.md | 39 -- docs/assays/metadata/GeoMx.md | 97 ---- docs/assays/metadata/HiFi.md | 47 -- docs/assays/metadata/Histology.md | 109 ++-- docs/assays/metadata/{testing => }/IMC-2D.md | 0 docs/assays/metadata/IMC.md | 234 -------- docs/assays/metadata/Illumina-Spatial.md | 41 -- docs/assays/metadata/LC-MS.md | 385 +++---------- .../metadata/{testing => }/Light-Sheet.md | 0 docs/assays/metadata/LightSheet.md | 146 ----- docs/assays/metadata/MALDI.md | 238 +++----- docs/assays/metadata/MERFISH.md | 47 -- docs/assays/metadata/MIBI.md | 190 +++---- docs/assays/metadata/MPLEx.md | 59 -- .../metadata/{testing => }/MUSIC-(CEDAR).md | 0 docs/assays/metadata/MUSIC.md | 128 +++-- docs/assays/metadata/Olink.md | 28 - docs/assays/metadata/PhenoCycler.md | 36 -- docs/assays/metadata/Pixel-seqV2.md | 37 -- .../{testing => }/RNAseq-(with-probes).md | 0 docs/assays/metadata/RNAseq.md | 520 ++++-------------- docs/assays/metadata/RNAseqWithProbes.md | 63 --- docs/assays/metadata/Raman-Imaging.md | 45 -- docs/assays/metadata/SIMS.md | 39 -- docs/assays/metadata/STARmap.md | 43 -- .../metadata/SecondHarmonicGeneration.md | 34 -- docs/assays/metadata/Seq-Scope.md | 39 -- docs/assays/metadata/Slide-seq.md | 94 ---- docs/assays/metadata/SnareSeq2.md | 19 - .../metadata/ThickSectionMultiphotonMxIF.md | 34 -- .../{testing => }/Visium-(no-probes).md | 0 docs/assays/metadata/Visium-HD.md | 33 -- docs/assays/metadata/VisiumNoProbes.md | 58 -- docs/assays/metadata/VisiumWithProbes.md | 30 - docs/assays/metadata/WGS.md | 86 --- docs/assays/metadata/{testing => }/comet.md | 0 .../{testing => }/cosmx-proteomics.md | 0 .../{testing => }/cosmx-transcriptomics.md | 0 docs/assays/metadata/{testing => }/cycif.md | 0 docs/assays/metadata/{testing => }/cytof.md | 0 .../metadata/{testing => }/dna-methylation.md | 0 .../metadata/{testing => }/enhancedsrs.md | 0 docs/assays/metadata/{testing => }/facs.md | 0 docs/assays/metadata/{testing => }/geomx.md | 0 docs/assays/metadata/{testing => }/hifi.md | 0 docs/assays/metadata/iCLAP.md | 34 -- docs/assays/metadata/{testing => }/iclap.md | 0 .../{testing => }/illumina-spatial.md | 0 docs/assays/metadata/{testing => }/imc.md | 0 docs/assays/metadata/index.md | 39 +- docs/assays/metadata/link2.png | Bin 1449 -> 0 bytes docs/assays/metadata/{testing => }/merfish.md | 0 docs/assays/metadata/{testing => }/mplex.md | 0 docs/assays/metadata/{testing => }/olink.md | 0 .../metadata/{testing => }/phenocycler.md | 0 .../metadata/{testing => }/pixel-seqv2.md | 0 .../metadata/{testing => }/raman-imaging.md | 0 .../{testing => }/secondharmonicgeneration.md | 0 .../metadata/{testing => }/seq-scope.md | 0 docs/assays/metadata/seqFISH.md | 90 --- docs/assays/metadata/{testing => }/seqfish.md | 0 docs/assays/metadata/{testing => }/simple.md | 0 docs/assays/metadata/{testing => }/sims.md | 0 .../metadata/{testing => }/slide-seq.md | 0 .../metadata/{testing => }/snareseq2.md | 0 docs/assays/metadata/{testing => }/starmap.md | 0 docs/assays/metadata/testing/4i.md | 21 - docs/assays/metadata/testing/ATACseq.md | 94 ---- docs/assays/metadata/testing/CODEX.md | 62 --- docs/assays/metadata/testing/DESI.md | 69 --- docs/assays/metadata/testing/Histology.md | 68 --- docs/assays/metadata/testing/LC-MS.md | 89 --- docs/assays/metadata/testing/MALDI.md | 67 --- docs/assays/metadata/testing/MIBI.md | 86 --- docs/assays/metadata/testing/MUSIC.md | 70 --- docs/assays/metadata/testing/RNAseq.md | 95 ---- docs/assays/metadata/testing/index.md | 49 -- docs/assays/metadata/testing/info3.png | Bin 2038 -> 0 bytes .../thicksectionmultiphotonmxif.md | 0 .../metadata/{testing => }/visium-hd.md | 0 .../{testing => }/visiumwithprobes.md | 0 docs/assays/metadata/{testing => }/wgs.md | 0 .../source/reharmonize-legacy-metadata | 1 + 100 files changed, 737 insertions(+), 4321 deletions(-) rename docs/assays/{metadata/testing => }/.directory (88%) rename docs/assays/metadata/{testing => }/10X-Multiome.md (100%) delete mode 100644 docs/assays/metadata/10XMultiome.md rename docs/assays/metadata/{testing => }/Auto-fluorescence.md (100%) delete mode 100644 docs/assays/metadata/AutoFluorescence.md delete mode 100644 docs/assays/metadata/COMET.md rename docs/assays/metadata/{testing => }/Cell-DIVE.md (100%) delete mode 100644 docs/assays/metadata/CosMx-Proteomics.md delete mode 100644 docs/assays/metadata/CosMx-Transcriptomics.md delete mode 100644 docs/assays/metadata/CyCIF.md delete mode 100644 docs/assays/metadata/CyTOF.md delete mode 100644 docs/assays/metadata/DNA-Methylation.md delete mode 100644 docs/assays/metadata/EnhancedSRS.md delete mode 100644 docs/assays/metadata/FACS.md delete mode 100644 docs/assays/metadata/GeoMx.md delete mode 100644 docs/assays/metadata/HiFi.md rename docs/assays/metadata/{testing => }/IMC-2D.md (100%) delete mode 100644 docs/assays/metadata/IMC.md delete mode 100644 docs/assays/metadata/Illumina-Spatial.md rename docs/assays/metadata/{testing => }/Light-Sheet.md (100%) delete mode 100644 docs/assays/metadata/LightSheet.md delete mode 100644 docs/assays/metadata/MERFISH.md delete mode 100644 docs/assays/metadata/MPLEx.md rename docs/assays/metadata/{testing => }/MUSIC-(CEDAR).md (100%) delete mode 100644 docs/assays/metadata/Olink.md delete mode 100644 docs/assays/metadata/PhenoCycler.md delete mode 100644 docs/assays/metadata/Pixel-seqV2.md rename docs/assays/metadata/{testing => }/RNAseq-(with-probes).md (100%) delete mode 100644 docs/assays/metadata/RNAseqWithProbes.md delete mode 100644 docs/assays/metadata/Raman-Imaging.md delete mode 100644 docs/assays/metadata/SIMS.md delete mode 100644 docs/assays/metadata/STARmap.md delete mode 100644 docs/assays/metadata/SecondHarmonicGeneration.md delete mode 100644 docs/assays/metadata/Seq-Scope.md delete mode 100644 docs/assays/metadata/Slide-seq.md delete mode 100644 docs/assays/metadata/SnareSeq2.md delete mode 100644 docs/assays/metadata/ThickSectionMultiphotonMxIF.md rename docs/assays/metadata/{testing => }/Visium-(no-probes).md (100%) delete mode 100644 docs/assays/metadata/Visium-HD.md delete mode 100644 docs/assays/metadata/VisiumNoProbes.md delete mode 100644 docs/assays/metadata/VisiumWithProbes.md delete mode 100644 docs/assays/metadata/WGS.md rename docs/assays/metadata/{testing => }/comet.md (100%) rename docs/assays/metadata/{testing => }/cosmx-proteomics.md (100%) rename docs/assays/metadata/{testing => }/cosmx-transcriptomics.md (100%) rename docs/assays/metadata/{testing => }/cycif.md (100%) rename docs/assays/metadata/{testing => }/cytof.md (100%) rename docs/assays/metadata/{testing => }/dna-methylation.md (100%) rename docs/assays/metadata/{testing => }/enhancedsrs.md (100%) rename docs/assays/metadata/{testing => }/facs.md (100%) rename docs/assays/metadata/{testing => }/geomx.md (100%) rename docs/assays/metadata/{testing => }/hifi.md (100%) delete mode 100644 docs/assays/metadata/iCLAP.md rename docs/assays/metadata/{testing => }/iclap.md (100%) rename docs/assays/metadata/{testing => }/illumina-spatial.md (100%) rename docs/assays/metadata/{testing => }/imc.md (100%) delete mode 100644 docs/assays/metadata/link2.png rename docs/assays/metadata/{testing => }/merfish.md (100%) rename docs/assays/metadata/{testing => }/mplex.md (100%) rename docs/assays/metadata/{testing => }/olink.md (100%) rename docs/assays/metadata/{testing => }/phenocycler.md (100%) rename docs/assays/metadata/{testing => }/pixel-seqv2.md (100%) rename docs/assays/metadata/{testing => }/raman-imaging.md (100%) rename docs/assays/metadata/{testing => }/secondharmonicgeneration.md (100%) rename docs/assays/metadata/{testing => }/seq-scope.md (100%) delete mode 100644 docs/assays/metadata/seqFISH.md rename docs/assays/metadata/{testing => }/seqfish.md (100%) rename docs/assays/metadata/{testing => }/simple.md (100%) rename docs/assays/metadata/{testing => }/sims.md (100%) rename docs/assays/metadata/{testing => }/slide-seq.md (100%) rename docs/assays/metadata/{testing => }/snareseq2.md (100%) rename docs/assays/metadata/{testing => }/starmap.md (100%) delete mode 100644 docs/assays/metadata/testing/4i.md delete mode 100644 docs/assays/metadata/testing/ATACseq.md delete mode 100644 docs/assays/metadata/testing/CODEX.md delete mode 100644 docs/assays/metadata/testing/DESI.md delete mode 100644 docs/assays/metadata/testing/Histology.md delete mode 100644 docs/assays/metadata/testing/LC-MS.md delete mode 100644 docs/assays/metadata/testing/MALDI.md delete mode 100644 docs/assays/metadata/testing/MIBI.md delete mode 100644 docs/assays/metadata/testing/MUSIC.md delete mode 100644 docs/assays/metadata/testing/RNAseq.md delete mode 100644 docs/assays/metadata/testing/index.md delete mode 100644 docs/assays/metadata/testing/info3.png rename docs/assays/metadata/{testing => }/thicksectionmultiphotonmxif.md (100%) rename docs/assays/metadata/{testing => }/visium-hd.md (100%) rename docs/assays/metadata/{testing => }/visiumwithprobes.md (100%) rename docs/assays/metadata/{testing => }/wgs.md (100%) create mode 160000 scripts/newMeta2/source/reharmonize-legacy-metadata diff --git a/docs/assays/metadata/testing/.directory b/docs/assays/.directory similarity index 88% rename from docs/assays/metadata/testing/.directory rename to docs/assays/.directory index 07af0c4e..7bf96541 100644 --- a/docs/assays/metadata/testing/.directory +++ b/docs/assays/.directory @@ -1,6 +1,6 @@ [Dolphin] HeaderColumnWidths=374,72,111,90,579,119,72 -Timestamp=2026,6,11,17,2,5.693 +Timestamp=2026,7,16,18,53,4.739 Version=4 ViewMode=1 VisibleRoles=Details_text,Details_size,Details_modificationtime,Details_type,Details_path,Details_creationtime,Details_owner,CustomizedDetails diff --git a/docs/assays/metadata/testing/10X-Multiome.md b/docs/assays/metadata/10X-Multiome.md similarity index 100% rename from docs/assays/metadata/testing/10X-Multiome.md rename to docs/assays/metadata/10X-Multiome.md diff --git a/docs/assays/metadata/10XMultiome.md b/docs/assays/metadata/10XMultiome.md deleted file mode 100644 index a1ab8b90..00000000 --- a/docs/assays/metadata/10XMultiome.md +++ /dev/null @@ -1,24 +0,0 @@ ---- -layout: page ---- -# 10X-Multiome - -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|----------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | True | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| True | -| number_of_pre-amplification_pcr_cycles | Numeric | The number of PCR cycles run after the Chromium Controller step and prior to separating the suspension and initiating library construction | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/4i.md b/docs/assays/metadata/4i.md index fe7bf3c1..ef03680f 100644 --- a/docs/assays/metadata/4i.md +++ b/docs/assays/metadata/4i.md @@ -1,37 +1,21 @@ ---- -layout: page --- -# 4i (Iterative Indirect Immunofluorescence Imaging) +layout: page-triary +--- + +# 4i Metadata Attributes -
Version 2 (current) +Fields that are collected for 4i data, available at ```dataset.metadata.``` +  -## Version 2 (current) +* indicates a required field -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| number_of_antibodies | Numeric | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | | True | -| number_of_biomarker_imaging_rounds | Numeric | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | | True | -| number_of_total_imaging_rounds | Numeric | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | | True | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| antibodies_path | Textfield | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | -| cell_boundary_marker_or_stain | Textfield | The name of the marker or stain used to identify all cell boundaries in the tissue. This name must exactly match the antibody-targeted molecule marker or non-antibody targeted molecule stain as found in the imaging data. For example, in the case of using the PhenoCycler, ensure the name corresponds to the value in the XPD output file. If multiple markers or stains are employed, list them in a comma-separated format. Example: Pan-Cytokeratin, E-Cadherin | | False | -| nuclear_marker_or_stain | Textfield | The nuclear marker or stain used, which can be an antibody-targeted molecule present in or around the cell nucleus. For protein targets, use the protein or gene symbol that identifies the antibody target, ensuring it matches the antibody target from the panel used or custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets, provide the stain name (e.g., DAPI) and, when applicable, include the associated staining kit and vendor. For the PhenoCycler, ensure the symbol matches the value found in the XPD output file. Example: DAPI | | False | -| number_of_channels | Numeric | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | True | +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| source_storage_duration_value * | | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | || time_since_acquisition_instrument_calibration_value | | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | +| contributors_path * | | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | || data_path * | | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | || number_of_antibodies * | | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | || number_of_biomarker_imaging_rounds * | | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | || number_of_total_imaging_rounds * | | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | || slide_id * | | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | || dataset_type * | | The specific type of dataset being produced. Example: RNAseq | ```Visium HD``` ```4i``` ```LC-MS``` ```Thick section Multiphoton MxIF``` ```Light Sheet``` ```ATACseq``` ```Resolve``` ```HiFi-Slide``` ```COMET``` ```MPLEx``` ```10X Multiome``` ```MALDI``` ```Histology``` ```Cell DIVE``` ```FACS``` ```MS Lipidomics``` ```Visium (no probes)``` ```MUSIC``` ```RNAseq``` ```GeoMx (NGS)``` ```GeoMx (nCounter)``` ```RNAseq (with probes)``` ```Singular Genomics G4X``` ```Molecular Cartography``` ```CosMx Transcriptomics``` ```MERFISH``` ```Pixel-seqV2``` ```2D Imaging Mass Cytometry``` ```Confocal``` ```seqFISH``` ```DART-FISH``` ```MIBI``` ```Olink``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```DESI``` ```Xenium``` ```CyCIF``` ```SNARE-seq2``` ```nanoSPLITS``` ```Stereo-seq``` ```Visium (with probes)``` ```SIMS``` ```Auto-fluorescence``` ```CyTOF``` ```CosMx Proteomics``` ```DBiT-seq``` ```PhenoCycler``` ```CODEX``` ```Second Harmonic Generation (SHG)``` ```Seq-Scope``` || analyte_class * | | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein``` ```Lipid + metabolite``` ```Collagen``` ```RNA``` ```Fluorochrome``` ```DNA``` ```Metabolite``` ```DNA + RNA``` ```Saturated lipid``` ```Lipid``` ```Peptide``` ```Protein``` ```Unsaturated lipid``` ```Endogenous fluorophore``` ```Chromatin``` ```Polysaccharide``` || acquisition_instrument_vendor * | | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics``` ```Cytek Biosciences``` ```Thermo Fisher Scientific``` ```Sciex``` ```Vizgen``` ```Leica Microsystems``` ```Akoya Biosciences``` ```Keyence``` ```Andor``` ```Standard BioTools (Fluidigm)``` ```Leica Biosystems``` ```Zeiss Microscopy``` ```Ionpath``` ```Motic``` ```In-House``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Element Biosciences``` ```Hamamatsu``` ```Bruker``` ```Illumina``` ```3DHISTECH``` ```Singular Genomics``` ```Huron Digital Pathology``` ```Resolve Biosciences``` ```NanoString``` ```Cytiva``` ```10x Genomics``` ```Microscopes International``` ```BGI Genomics``` || acquisition_instrument_model * | | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X``` ```NovaSeq X Plus``` ```Cytek Northern Lights``` ```Lightsheet 7``` ```Resolve Biosciences Molecular Cartography``` ```timsTOF HT``` ```timsTOF Pro 2``` ```timsTOF Pro``` ```timsTOF Ultra``` ```timsTOF Ultra 2``` ```timsTOF SCP``` ```Axio Scan.Z1``` ```MALDI timsTOF Flex Prototype``` ```CosMx Spatial Molecular Imager``` ```Unknown``` ```MERSCOPE Ultra``` ```Juno System``` ```timsTOF FleX``` ```Custom: Multiphoton``` ```CyTOF XT``` ```Helios``` ```EVOS M7000``` ```Aperio AT2``` ```Phenocycler-Fusion 2.0``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Observer 3``` ```NanoZoomer-SQ``` ```NanoZoomer S210``` ```NanoZoomer S60``` ```NanoZoomer S360``` ```DM6 B``` ```MoticEasyScan One``` ```In-House``` ```NextSeq 500``` ```BZ-X710``` ```QTRAP 5500``` ```NextSeq 550``` ```HiSeq 2500``` ```HiSeq 4000``` ```NovaSeq 6000``` ```Q Exactive HF``` ```Orbitrap Fusion Lumos Tribrid``` ```Q Exactive``` ```VS200 Slide Scanner``` ```Not applicable``` ```Orbitrap Eclipse Tribrid``` ```MIBIscope``` ```IN Cell Analyzer 2200``` ```timsTOF FleX MALDI-2``` || source_storage_duration_unit * | | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour``` ```month``` ```day``` ```minute``` ```year``` || time_since_acquisition_instrument_calibration_unit | | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month``` ```day``` ```year``` | +| metadata_schema_id * | | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | || preparation_protocol_doi * | | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | || is_targeted * | | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | || antibodies_path * | | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | || parent_sample_id * | | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | || non_global_files | | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | +| cell_boundary_marker_or_stain | | The name of the marker or stain used to identify all cell boundaries in the tissue. This name must exactly match the antibody-targeted molecule marker or non-antibody targeted molecule stain as found in the imaging data. For example, in the case of using the PhenoCycler, ensure the name corresponds to the value in the XPD output file. If multiple markers or stains are employed, list them in a comma-separated format. Example: Pan-Cytokeratin, E-Cadherin | | +| nuclear_marker_or_stain | | The nuclear marker or stain used, which can be an antibody-targeted molecule present in or around the cell nucleus. For protein targets, use the protein or gene symbol that identifies the antibody target, ensuring it matches the antibody target from the panel used or custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets, provide the stain name (e.g., DAPI) and, when applicable, include the associated staining kit and vendor. For the PhenoCycler, ensure the symbol matches the value found in the XPD output file. Example: DAPI | | +| number_of_channels * | | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | -
\ No newline at end of file diff --git a/docs/assays/metadata/ATACseq.md b/docs/assays/metadata/ATACseq.md index 3e030c10..f83141cb 100644 --- a/docs/assays/metadata/ATACseq.md +++ b/docs/assays/metadata/ATACseq.md @@ -1,257 +1,94 @@ ---- -layout: page ---- -# ATACseq - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - - -
Version 3 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | False | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | False | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina``` ```HiSeq 3000/4000 PE Cluster Kit PE-410-1001``` ```PN 1000283, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles)``` ```PN 20046811, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles)``` ```PN 20046812, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles)``` ```PN 20046813, Illumina``` ```NextSeq 2000 P3 Reagent Kit (300 Cycles)``` ```PN 20040561, Illumina``` ```NextSeq 2000 P3 Reagents Kit (100 Cycles)``` ```PN 20040559, Illumina``` ```NextSeq 500/550 Hi Output Kit 150 Cycles``` ```v2.5``` ```PN 20024907, Illumina``` ```NextSeq 500/550 Hi Output Kit 75 Cycles v2.5``` ```PN 20024906, Illumina``` ```NextSeq 500/550 Mid Output Kit 150 Cycles v2.5``` ```PN 20024904, Illumina``` ```NovaSeq 6000 S1 Reagent Kit (200 Cycles)``` ```PN 20012864, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028319, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028318, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles)``` ```PN 20028317, Illumina``` ```NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028316, Illumina``` ```NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles)``` ```PN 20028312, Illumina``` ```NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028313, Illumina``` ```NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles)``` ```PN 20028401, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (100 Cycle)``` ```PN 20104703, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (200 Cycle)``` ```PN 20104704, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (300 Cycle)``` ```PN 20104705, Illumina``` ```NovaSeq X Series 10B Reagent Kit (100 Cycle)``` ```PN 20085596, Illumina``` ```NovaSeq X Series 10B Reagent Kit (200 Cycle)``` ```PN 20085595, Illumina``` ```NovaSeq X Series 10B Reagent Kit (300 Cycle) ``` ```PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| transposition_method | Allowable Value | Modality of capturing accessible chromatin molecules. For example, this would be the type of kit that was used. | ```bulkATACseq``` ```sciATACseq``` ```Custom``` ```scATACseq``` | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | - -
- -
SNARE-seq2 / sciATACseq / snATACseq Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['SNARE-seq2', 'sciATACseq', 'snATACseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| is_technical_replicate | boolean | If TRUE, fastq files in dataset need to be merged. | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| sc_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol. | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. "OK" or "not OK", or with more specificity such as "debris", "clump", "low clump". | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment. | | True | -| transposition_input | Numeric | Number of cell/nuclei input to the assay. | | True | -| transposition_method | Allowable Value | Modality of capturing accessible chromatin molecules. | ['SNARE-Seq2-AC', 'bulkATACseq', 'snATACseq', 'sciATACseq'] | True | -| transposition_transposase_source | Allowable Value | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | ['10X snATAC', 'In-house', 'Nextera', '10X multiome'] | True | -| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". DOI for protocols.io referring to the protocol for this assay. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming. | | True | -| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). This field is not required for barcoding by single-cell combinatorial indexing. | | False | -| cell_barcode_offset | Textfield | Positions in the read at which the cell barcodes start. Cell barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. (Does not apply to sciATACseq, SNARE-seq and BulkATAC.) | | False | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs. Cell barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. (Does not apply to sciATACseq, SNARE-seq and BulkATAC.) | | False | -| library_pcr_cycles | Numeric | Number of PCR cycles to enrich for accessible chromatin fragments. | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library generation (figure in Descriptions section) | | True | -| library_final_yield | Numeric | Total ng of library after final pcr amplification step. | | True | -| library_final_yield_unit | Allowable Value | Units for library_final_yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
SNARE-seq2 / sciATACseq / snATACseq Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Textfield | The type of single cell entity derived from isolation protocol | | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Textfield | The method by which specific cell populations are sorted or enriched. | | False | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | boolean | Is the sequencing reaction run in repliucate, TRUE or FALSE | | True | -| cell_barcode_read | Textfield | Which read file contains the cell barcode | | True | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | True | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | True | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
bulkATACseq Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['bulkATACseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| bulk_transposition_input_number_nuclei | Textfield | A number (no comma separators) | | True | -| bulk_atac_cell_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | True | -| is_technical_replicate | boolean | Is this a sequencing replicate? | | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library_concentration_value | ['nM'] | False | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_creation_date | Datetime | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | False | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_pcr_cycles | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | True | -| library_preparation_kit | Textfield | Reagent kit used for library preparation | | True | -| sample_quality_metric | Textfield | This is a quality metric by visual inspection. This should answerthe question: Are the nuclei intact and are the nuclei free of significant amountsof debris? This can be captured at a high level, “OK” or “notOK”. | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | -| transposition_method | Textfield | Modality of capturing accessible chromatin molecules. The kit used, for example. | | True | -| transposition_transposase_source | Textfield | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- - - -
bulkATACseq 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['bulkATACseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| bulk_transposition_input_number_nuclei | Textfield | A number (no comma separators) | | True | -| bulk_atac_cell_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | True | -| is_technical_replicate | boolean | Is this a sequencing replicate? | | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library_concentration_value | ['nM'] | False | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_creation_date | Datetime | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | False | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_pcr_cycles | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | True | -| library_preparation_kit | Textfield | Reagent kit used for library preparation | | True | -| sample_quality_metric | Textfield | This is a quality metric by visual inspection. This should answerthe question: Are the nuclei intact and are the nuclei free of significant amountsof debris? This can be captured at a high level, “OK” or “notOK”. | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | -| transposition_method | Textfield | Modality of capturing accessible chromatin molecules. The kit used, for example. | | True | -| transposition_transposase_source | Textfield | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
+--- +layout: page-triary +--- + +# ATACseq Metadata Attributes + +Fields that are collected for ATACseq data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| barcode_offset *| | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | +| barcode_read *| | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | +| barcode_size *| | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | +| umi_offset *| | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | +| umi_read *| | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | +| umi_size *| | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | +| assay_input_entity *| | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | +| number_of_input_cells_or_nuclei | | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | +| library_adapter_sequence *| | Adapter sequence to be used for adapter trimming | | +| library_average_fragment_size *| | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | +| library_input_amount_value | | The amount of cDNA, after amplification, that was used for library construction. | | +| library_input_amount_unit | | unit of library input amount value | ```ng``` ```ul``` | +| library_output_amount_value | | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | +| library_output_amount_unit | | Units of library final yield. | ```ng``` ```ul``` | +| library_concentration_value *| | The concentration value of the pooled library samples submitted for sequencing. | | +| library_concentration_unit *| | Unit of library_concentration_value | ```ng/ul``` ```nM``` | +| library_layout *| | State whether the library was generated for single-end or paired end sequencing. | ```paired-end``` ```single-end``` | +| number_of_pcr_cycles_for_indexing *| | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | +| library_preparation_kit *| | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```1 slides``` ```4 reactions; PN 1000338``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 reactions; PN 1000187``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | +| sample_indexing_kit *| | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001``` | +| sample_indexing_set *| | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | +| is_technical_replicate *| | Is this a sequencing replicate? | | +| expected_entity_capture_count | | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | +| sequencing_reagent_kit *| | Reagent kit used for sequencing | ```Custom``` ```Illumina``` ```HiSeq 3000/4000 PE Cluster Kit PE-410-1001``` ```PN 1000283, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles)``` ```PN 20046811, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles)``` ```PN 20046812, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles)``` ```PN 20046813, Illumina``` ```NextSeq 2000 P3 Reagent Kit (300 Cycles)``` ```PN 20040561, Illumina``` ```NextSeq 2000 P3 Reagents Kit (100 Cycles)``` ```PN 20040559, Illumina``` ```NextSeq 500/550 Hi Output Kit 150 Cycles``` ```v2.5``` ```PN 20024907, Illumina``` ```NextSeq 500/550 Hi Output Kit 75 Cycles v2.5``` ```PN 20024906, Illumina``` ```NextSeq 500/550 Mid Output Kit 150 Cycles v2.5``` ```PN 20024904, Illumina``` ```NovaSeq 6000 S1 Reagent Kit (200 Cycles)``` ```PN 20012864, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028319, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028318, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles)``` ```PN 20028317, Illumina``` ```NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028316, Illumina``` ```NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles)``` ```PN 20028312, Illumina``` ```NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028313, Illumina``` ```NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles)``` ```PN 20028401, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (100 Cycle)``` ```PN 20104703, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (200 Cycle)``` ```PN 20104704, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (300 Cycle)``` ```PN 20104705, Illumina``` ```NovaSeq X Series 10B Reagent Kit (100 Cycle)``` ```PN 20085596, Illumina``` ```NovaSeq X Series 10B Reagent Kit (200 Cycle)``` ```PN 20085595, Illumina``` ```NovaSeq X Series 10B Reagent Kit (300 Cycle)``` ```PN 20085594``` | +| sequencing_read_format *| | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | +| transposition_reagent_kit | | | | +| transposition_method *| | Modality of capturing accessible chromatin molecules. The kit used, for example. | ```bulkATACseq``` ```sciATACseq``` ```Custom``` ```scATACseq``` | +| sequencing_batch_id | | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| capture_batch_id | | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_instrument_kit | | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| library_construction_protocols_io_doi | | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | +| library_creation_date | | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | +| library_id | | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| sample_quality_metric | | This is a quality metric by visual inspection. This should answer the question: Are the nuclei intact and are the nuclei free of significant amounts of debris? This can be captured at a high level, “OK” or “not OK”. | | +| library_pcr_cycles | | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | +| bulk_atac_cell_isolation_protocols_io_doi | | Link to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | +| sc_isolation_enrichment | | The method by which specific cell populations are sorted or enriched. | ```none``` ```FACS``` | +| sc_isolation_protocols_io_doi | | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | +| sc_isolation_quality_metric | | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. "OK" or "not OK", or with more specificity such as "debris", "clump", "low clump". | | +| sc_isolation_tissue_dissociation | | The method by which tissues are dissociated into single cells in suspension. | | +| sc_isolation_cell_number | | Total number of cell/nuclei yielded post dissociation and enrichment. | | +| sequencing_phix_percent | | Percent PhiX loaded to the run | | +| sequencing_read_percent_q30 | | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | +| transposition_transposase_source | | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | ```10X snATAC``` ```In-house``` ```Nextera``` ```10X multiome``` | +| version | | Version of the schema to use when validating this metadata. | ```1``` | +| description | | Free-text description of this assay. | | diff --git a/docs/assays/metadata/testing/Auto-fluorescence.md b/docs/assays/metadata/Auto-fluorescence.md similarity index 100% rename from docs/assays/metadata/testing/Auto-fluorescence.md rename to docs/assays/metadata/Auto-fluorescence.md diff --git a/docs/assays/metadata/AutoFluorescence.md b/docs/assays/metadata/AutoFluorescence.md deleted file mode 100644 index 6abca683..00000000 --- a/docs/assays/metadata/AutoFluorescence.md +++ /dev/null @@ -1,105 +0,0 @@ ---- -layout: page ---- -# Auto-fluorescence - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 2 (Latest) - -## Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement | ```month``` ```day``` ```year``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
Version 1 - -## Version 1 - -| Attribute | Type | Description | AllowableValues | Required | -|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['AF'] | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that this is resolution within a given sample: z-pitch (resolution_z_value) is the increment distance between image slices, ie. the microscope stage is moved up or down in increments to capture images of several focal planes. | | True | -| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | -| number_of_channels | Numeric | Number of channels capturing the emission spectrum from natural fluorophores in the sample. | | True | -| overall_protocols_io_doi | Textfield | DOI for protocols.io referring to the overall protocol for the assay. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | AllowableValues | Required | -|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['AF'] | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that this is resolution within a given sample: z-pitch (resolution_z_value) is the increment distance between image slices, ie. the microscope stage is moved up or down in increments to capture images of several focal planes. | | True | -| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | -| number_of_channels | Numeric | Number of channels capturing the emission spectrum from natural fluorophores in the sample. | | True | -| overall_protocols_io_doi | Textfield | DOI for protocols.io referring to the overall protocol for the assay. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/CODEX.md b/docs/assays/metadata/CODEX.md index f2efa40f..4550574c 100644 --- a/docs/assays/metadata/CODEX.md +++ b/docs/assays/metadata/CODEX.md @@ -1,120 +1,62 @@ ---- -layout: page ---- -# CODEX - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 2 (Latest) - -## Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| number_of_biomarker_imaging_rounds | Numeric | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo application, 2. fluor application, 3. washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | True | -| number_of_total_imaging_rounds | Numeric | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| total_run_time_value | Numeric | How long the tissue was on the acquisition instrument. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| total_run_time_unit | Allowable Value | The units for the total run time unit field. | ```Hour``` ```Minute``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | True | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
Version 1 - -## Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['CODEX', 'CODEX2'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes', 'No'] | True | -| acquisition_instrument_vendor | Allowable Value | An acquisition_instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing molecular mass. | ['Keyence', 'Zeiss'] | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | ['BZ-X800', 'BZ-X710', 'Axio Observer Z1'] | True | -| resolution_x_value | Numeric | The width of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of width of a pixel.(nm) | ['mm', 'um', 'nm'] | False | -| resolution_y_value | Numeric | The height of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of height of a pixel. (nm) | ['mm', 'um', 'nm'] | False | -| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that thisis resolution within a given sample: z-pitch (resolution_z_value) is the incrementdistance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stageis moved up or down in increments of 1.5um to capture images of several focalplanes. The best one will be used & the rest discarded. The thickness of the sampleitself is sample metadata. | | False | -| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare the sample for theassay. | ['CODEX'] | True | -| preparation_instrument_model | Allowable Value | The model number/name of the instrument used to prepare the samplefor the assay | ['version 1 robot', 'prototype robot - Stanford/Nolan Lab'] | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| number_of_cycles | Numeric | Number of cycles of 1. oligo application, 2. fluor application, 3.washes | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagentsfor the assay. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['CODEX'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes', 'No'] | True | -| acquisition_instrument_vendor | Allowable Value | An acquisition_instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing molecular mass. | ['Keyence', 'Zeiss'] | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | ['BZ-X800', 'BZ-X710', 'Axio Observer Z1'] | True | -| resolution_x_value | Numeric | The width of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of width of a pixel.(nm) | ['mm', 'um', 'nm'] | False | -| resolution_y_value | Numeric | The height of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of height of a pixel. (nm) | ['mm', 'um', 'nm'] | False | -| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that thisis resolution within a given sample: z-pitch (resolution_z_value) is the incrementdistance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stageis moved up or down in increments of 1.5um to capture images of several focalplanes. The best one will be used & the rest discarded. The thickness of the sampleitself is sample metadata. | | False | -| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | -| | Textfield | | | | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare the sample for theassay. | ['CODEX'] | True | -| preparation_instrument_model | Allowable Value | The model number/name of the instrument used to prepare the samplefor the assay | ['version 1 robot', 'prototype robot - Stanford/Nolan Lab'] | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| number_of_cycles | Numeric | Number of cycles of 1. oligo application, 2. fluor application, 3.washes | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagentsfor the assay. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
+--- +layout: page-triary +--- + +# CODEX Metadata Attributes + +Fields that are collected for CODEX data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor *| | An acquisition_instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| antibodies_path *| | Relative path to file with antibody information for this dataset. | | +| preparation_instrument_vendor *| | The manufacturer of the instrument used to prepare the sample for the assay. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model *| | The model number/name of the instrument used to prepare the sample for the assay | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| total_run_time_value | | How long the tissue was on the acquisition instrument. | | +| total_run_time_unit | | The units for the total run time unit field. | ```Hour``` ```Minute``` | +| number_of_antibodies *| | Number of antibodies | | +| number_of_channels *| | Number of fluorescent channels imaged during each cycle. | | +| number_of_biomarker_imaging_rounds *| | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo application, 2. fluor application, 3. washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | +| number_of_total_imaging_rounds *| | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | +| slide_id | | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| resolution_z_unit | | The unit of incremental distance between image slices. | ```mm``` ```um``` ```nm``` | +| resolution_z_value | | Optional if assay does not have multiple z-levels. Note that this is resolution within a given sample: z-pitch (resolution_z_value) is the increment distance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stage is moved up or down in increments of 1.5um to capture images of several focal planes. The best one will be used & the rest discarded. The thickness of the sample itself is sample metadata. | | diff --git a/docs/assays/metadata/COMET.md b/docs/assays/metadata/COMET.md deleted file mode 100644 index 7a7348b1..00000000 --- a/docs/assays/metadata/COMET.md +++ /dev/null @@ -1,37 +0,0 @@ ---- -layout: page ---- -# COMET - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| number_of_antibodies | Numeric | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | | True | -| number_of_biomarker_imaging_rounds | Numeric | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | | True | -| number_of_total_imaging_rounds | Numeric | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | | True | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| antibodies_path | Textfield | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | -| cell_boundary_marker_or_stain | Textfield | The name of the marker or stain used to identify all cell boundaries in the tissue. This name must exactly match the antibody-targeted molecule marker or non-antibody targeted molecule stain as found in the imaging data. For example, in the case of using the PhenoCycler, ensure the name corresponds to the value in the XPD output file. If multiple markers or stains are employed, list them in a comma-separated format. Example: Pan-Cytokeratin, E-Cadherin | | False | -| nuclear_marker_or_stain | Textfield | The nuclear marker or stain used, which can be an antibody-targeted molecule present in or around the cell nucleus. For protein targets, use the protein or gene symbol that identifies the antibody target, ensuring it matches the antibody target from the panel used or custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets, provide the stain name (e.g., DAPI) and, when applicable, include the associated staining kit and vendor. For the PhenoCycler, ensure the symbol matches the value found in the XPD output file. Example: DAPI | | False | -| number_of_channels | Numeric | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/testing/Cell-DIVE.md b/docs/assays/metadata/Cell-DIVE.md similarity index 100% rename from docs/assays/metadata/testing/Cell-DIVE.md rename to docs/assays/metadata/Cell-DIVE.md diff --git a/docs/assays/metadata/CosMx-Proteomics.md b/docs/assays/metadata/CosMx-Proteomics.md deleted file mode 100644 index d50cdc02..00000000 --- a/docs/assays/metadata/CosMx-Proteomics.md +++ /dev/null @@ -1,43 +0,0 @@ ---- -layout: page ---- -# CosMx Proteomics - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | False | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | False | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | -| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | -| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | -| oligo_probe_panel | Assigned Value | The oligo probe panel used to target genes and/or proteins. If there is a core panel along with add-on modules, the core panel should be selected in this field. Any additional panels utilized should be documented in the "additional_panels_used.csv" file, which must be uploaded alongside the dataset. Example: 10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363 | ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-MsWTA-4```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (RNA, 1000 Plex); PN CMX-M-NEUP-R```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 16 rxns; PN 1000420```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 4 rxns x 4 BC; PN 1000475```, ```NanoString Technologies; CosMx Human 6K Discovery Panel (RNA, 6175 Plex); PN 121500041```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome, 4 rxns x 1 BC; PN 1000474```, ```10x Genomics; Xenium Human Multi-Tissue and Cancer Panel v1; PN 1000626```, ```NanoString Technologies; CosMx Human Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-H-USCP-1KP-R```, ```10x Genomics; Xenium Custom Gene Expression Panel (up to 50 genes); PN 1000464```, ```NanoString Technologies; CosMx Hs Univ Cell (RNA, 1000 Plex); PN 121500002```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365```, ```10x Genomics; Xenium Mouse Multi-Tissue Atlassing Panel; PN 1000627```, ```10x Genomics; Xenium Custom Gene Expression Panel (51-100 genes); PN 1000561```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 64 rxns; PN 1000456```, ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-HuWTA-4```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit v2.0 - Small; PN 1000667```, ```NanoString Technologies; CosMx Mouse Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-M-USCP-1KP-R```, ```NanoString Technologies; CosMx Human Immuno-Oncology Panel (Protein, 64 Plex); PN CMX-H-IOP-64P-P```, ```Custom```, ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466```, ```10x Genomics; Xenium Human Colon Gene Expression Panel; PN 1000642```, ```10x Genomics; Chromium Next GEM Single Cell Fixed RNA Mouse Transcriptome Probe Kit, 64 rxns; PN 1000492```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (Protein, 64 Plex); PN CMX-M-Neuro-64P-P```, ```NanoString Technologies; CosMx Hs WTX RNA Panel Kit, 2 slides: PN 121500047```, ```10x Genomics; Xenium Human Lung Gene Expression Panel; PN 1000601```, ```10x Genomics; Xenium Prime 5K Human Pan Tissue & Pathways Panel; PN 1000724``` | False | -| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | -| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | -| roi_label | Textfield | The label for the region of interest (ROI). For Resolve and CosMx, this corresponds to the field of view (FOV) label. In the case of Xenium, it refers to the ID of the region containing the analysis. For GeoMx, this information can be located in the "Initial Dataset" spreadsheet, which can be downloaded from within the Data Analysis Suite. Example: Decidua | | False | -| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | -| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/CosMx-Transcriptomics.md b/docs/assays/metadata/CosMx-Transcriptomics.md deleted file mode 100644 index 62695999..00000000 --- a/docs/assays/metadata/CosMx-Transcriptomics.md +++ /dev/null @@ -1,88 +0,0 @@ ---- -layout: page ---- -# CosMx Transcriptomics - -
Version 3 (current) - -## Version 3 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial v0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | False | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | False | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | -| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| proteinasek_concentration | Numeric | The concentration of the enzyme Proteinase K within a sample, measured in micrograms per milliliter (ug/ml). Example: 10 | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is incubated with Proteinase K. Example: 15 | | False | -| proteinasek_incubation_time_unit | Assigned Value | The unit of measurement for the proteinaseK incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | -| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | -| oligo_probe_panel | Assigned Value | The oligo probe panel used to target genes and/or proteins. If there is a core panel along with add-on modules, the core panel should be selected in this field. Any additional panels utilized should be documented in the "additional_panels_used.csv" file, which must be uploaded alongside the dataset. Example: 10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363 | ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-MsWTA-4```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 16 rxns x 16 BC; PN 1000547```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (RNA, 1000 Plex); PN CMX-M-NEUP-R```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 16 rxns; PN 1000420```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 4 rxns x 4 BC; PN 1000475```, ```NanoString Technologies; CosMx Human 6K Discovery Panel (RNA, 6175 Plex); PN 121500041```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome, 4 rxns x 1 BC; PN 1000474```, ```10x Genomics; Xenium Human Multi-Tissue and Cancer Panel v1; PN 1000626```, ```NanoString Technologies; CosMx Human Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-H-USCP-1KP-R```, ```10x Genomics; GEM-X Flex Human Transcriptome Probe Kit, 16 samples; PN 1000785```, ```10x Genomics; Xenium Custom Gene Expression Panel (up to 50 genes); PN 1000464```, ```NanoString Technologies; CosMx Hs Univ Cell (RNA, 1000 Plex); PN 121500002```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365```, ```10x Genomics; Xenium Mouse Multi-Tissue Atlassing Panel; PN 1000627```, ```10x Genomics; Xenium Custom Gene Expression Panel (51-100 genes); PN 1000561```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 64 rxns; PN 1000456```, ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-HuWTA-4```, ```NanoString Technologies; GeoMx Human IO Proteome Atlas, 4 slides; PN 121300160```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit v2.0 - Small; PN 1000667```, ```NanoString Technologies; CosMx Mouse Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-M-USCP-1KP-R```, ```NanoString Technologies; CosMx Human Immuno-Oncology Panel (Protein, 64 Plex); PN CMX-H-IOP-64P-P```, ```Custom```, ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466```, ```10x Genomics; Xenium Human Colon Gene Expression Panel; PN 1000642```, ```10x Genomics; Chromium Next GEM Single Cell Fixed RNA Mouse Transcriptome Probe Kit, 64 rxns; PN 1000492```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (Protein, 64 Plex); PN CMX-M-Neuro-64P-P```, ```NanoString Technologies; CosMx Hs WTX RNA Panel Kit, 2 slides: PN 121500047```, ```10x Genomics; Xenium Human Lung Gene Expression Panel; PN 1000601```, ```10x Genomics; Xenium Prime 5K Human Pan Tissue & Pathways Panel; PN 1000724``` | True | -| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | -| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | -| roi_label | Textfield | The label for the region of interest (ROI). For Resolve and CosMx, this corresponds to the field of view (FOV) label. In the case of Xenium, it refers to the ID of the region containing the analysis. For GeoMx, this information can be located in the "Initial Dataset" spreadsheet, which can be downloaded from within the Data Analysis Suite. Example: Decidua | | False | -| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | -| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | - -
- - -
Version 2 - -## Version 2 - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial v0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | False | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | False | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | -| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| proteinasek_concentration | Numeric | The concentration of the enzyme Proteinase K within a sample, measured in micrograms per milliliter (ug/ml). Example: 10 | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is incubated with Proteinase K. Example: 15 | | False | -| proteinasek_incubation_time_unit | Assigned Value | The unit of measurement for the proteinaseK incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | -| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | -| oligo_probe_panel | Assigned Value | The oligo probe panel used to target genes and/or proteins. If there is a core panel along with add-on modules, the core panel should be selected in this field. Any additional panels utilized should be documented in the "additional_panels_used.csv" file, which must be uploaded alongside the dataset. Example: 10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363 | ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-MsWTA-4```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 16 rxns x 16 BC; PN 1000547```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (RNA, 1000 Plex); PN CMX-M-NEUP-R```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 16 rxns; PN 1000420```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 4 rxns x 4 BC; PN 1000475```, ```NanoString Technologies; CosMx Human 6K Discovery Panel (RNA, 6175 Plex); PN 121500041```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome, 4 rxns x 1 BC; PN 1000474```, ```10x Genomics; Xenium Human Multi-Tissue and Cancer Panel v1; PN 1000626```, ```NanoString Technologies; CosMx Human Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-H-USCP-1KP-R```, ```10x Genomics; GEM-X Flex Human Transcriptome Probe Kit, 16 samples; PN 1000785```, ```10x Genomics; Xenium Custom Gene Expression Panel (up to 50 genes); PN 1000464```, ```NanoString Technologies; CosMx Hs Univ Cell (RNA, 1000 Plex); PN 121500002```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365```, ```10x Genomics; Xenium Mouse Multi-Tissue Atlassing Panel; PN 1000627```, ```10x Genomics; Xenium Custom Gene Expression Panel (51-100 genes); PN 1000561```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 64 rxns; PN 1000456```, ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-HuWTA-4```, ```NanoString Technologies; GeoMx Human IO Proteome Atlas, 4 slides; PN 121300160```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit v2.0 - Small; PN 1000667```, ```NanoString Technologies; CosMx Mouse Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-M-USCP-1KP-R```, ```NanoString Technologies; CosMx Human Immuno-Oncology Panel (Protein, 64 Plex); PN CMX-H-IOP-64P-P```, ```Custom```, ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466```, ```10x Genomics; Xenium Human Colon Gene Expression Panel; PN 1000642```, ```10x Genomics; Chromium Next GEM Single Cell Fixed RNA Mouse Transcriptome Probe Kit, 64 rxns; PN 1000492```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (Protein, 64 Plex); PN CMX-M-Neuro-64P-P```, ```NanoString Technologies; CosMx Hs WTX RNA Panel Kit, 2 slides: PN 121500047```, ```10x Genomics; Xenium Human Lung Gene Expression Panel; PN 1000601```, ```10x Genomics; Xenium Prime 5K Human Pan Tissue & Pathways Panel; PN 1000724``` | True | -| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | -| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | -| roi_label | Textfield | The label for the region of interest (ROI). For Resolve and CosMx, this corresponds to the field of view (FOV) label. In the case of Xenium, it refers to the ID of the region containing the analysis. For GeoMx, this information can be located in the "Initial Dataset" spreadsheet, which can be downloaded from within the Data Analysis Suite. Example: Decidua | | False | -| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | -| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/CyCIF.md b/docs/assays/metadata/CyCIF.md deleted file mode 100644 index 2d293e15..00000000 --- a/docs/assays/metadata/CyCIF.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# CyCIF - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_biomarker_imaging_rounds | Numeric | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo and fluor application, 2. imaging, 3. removal of oligo and fluor along washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | True | -| number_of_total_imaging_rounds | Numeric | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/CyTOF.md b/docs/assays/metadata/CyTOF.md deleted file mode 100644 index f7f95499..00000000 --- a/docs/assays/metadata/CyTOF.md +++ /dev/null @@ -1,38 +0,0 @@ ---- -layout: page ---- -# CyTOF - -
Version 2 (Latest) - -## Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|----------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------|----------| -| lab_id | Textfield | An internal attribute labs can use to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This attribute will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Textfield | The specific type of dataset being produced. | | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Textfield | The time duration unit of measurement | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Textfield | The time unit of measurement | | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: [https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1](https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1). | | True | -| is_targeted | Assigned Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Custom``` ```None``` ```Sigma Aldrich; Cisplatin 25mg; PN P4394``` ```Standard BioTools; Cell-ID Cisplatin-198Pt 100 uL; PN 201198``` ```Standard BioTools; Cell-ID Intercalator-103Rh 2,000 um; PN 201103B``` ```Standard BioTools; Cell-ID Cisplatin-196Pt 100 uL; PN 201196``` ```Standard BioTools; Cell-ID Cisplatin 100 uL; PN 201064``` ```Standard BioTools; Cell-ID Intercalator-103Rh 500 um; PN 201103A``` ```Standard BioTools; Cell-ID Cisplatin-194Pt 100 uL; PN 201194``` ```Standard BioTools; Cell-ID Cisplatin-195Pt 100 uL; PN 201195``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| number_of_mass_channels | Numeric | The number of mass channels that measure the expression of markers in single cells. | | False | -| is_erythrocyte_lysis_performed | Assigned Value | Process in which red blood cells (RBCs) are broken down in the sample prior to analysis, thereby allowing researchers to focus primarily on white blood cells (WBCs). | ```Yes``` ```No``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | False | -| antibody_reagent_kit | Textfield | The kit containing the set of antibodies pre-conjugated with different heavy metal isotopes used to simultaneously detect and quantify multiple protein markers on individual cells by attaching these metal-labeled antibodies to specific cellular targets, essentially acting as the key component for labeling cells with the various markers needed for analysis on the CyTOF machine. | | False | -| viability_reagent_kit | Textfield | The kit used to differentiate between live and dead cells within a sample by selectively staining dead cells with a dye that can be detected by the instrument, allowing researchers to exclude dead cell data from their analysis and ensure accurate results when studying cell populations. | | False | -| is_cell_activation_performed | Assigned Value | Process by which ligand is binded to its receptors on a cell, which enhances the cell's ability to respond to various stimuli. | ```Yes``` ```No```e | False | -| activation_stimulus | Textfield | Specific type of stimulus used to provoke cell activation. Examples would include PMA/ionomycin or CD28in/brefeldin A. This field is required if "is_cells_activation performed" is Yes. | | False | -| is_fcr_blocking_applied | Assigned Value | Process by which a reagent has been added to the staining procedure to block the binding of antibodies to Fc receptors (FcRs) on cells, preventing non-specific binding and ensuring that only the intended target antigen is detected by the antibodies; essentially, it helps to minimize false positive signals by preventing antibodies from attaching to the cell via their Fc region instead of the antigen-specific binding site. | ```Yes``` ```No``` | False | -| is_heparin_used | Assigned Value | Indicates whether heparin was used ("Yes") or not ("No") during staining to prevent non-specific binding of metal-labeled antibodies to eosinophils to reduce background noise. |```Yes``` ```No``` | False | -| loaded_cell_concentration_value | Numeric | The number of cells present within a given volume of liquid for the experiment immediately prior to the experiment, essentially indicating how densely packed the cells are in a solution. | | False | -| loaded_cell_concentration_unit | Textfield | Unit of measure for cell concentration, e.g. cells per milliliter (cells/mL). | | False | -| instrument_calibration_bead_kit | Textfield | A set of beads of known mass intensity used to adjust the settings of a flow cytometer to ensure accurate measurements. | | False | -| calibration_kit_lot_number | Textfield | Manufacturer's lot number for the calibration bead kit used for the experiment. | | False | diff --git a/docs/assays/metadata/DESI.md b/docs/assays/metadata/DESI.md index 7ead4ea6..c651b8f3 100644 --- a/docs/assays/metadata/DESI.md +++ b/docs/assays/metadata/DESI.md @@ -1,43 +1,69 @@ ---- -layout: page ---- -# DESI - -
Version 2 (Latest) - -## Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | -| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | True | -| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | True | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | True | -| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | -| matrix_deposition_method | Allowable Value | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| preparation_matrix | Allowable Value | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| ion_mobility | Allowable Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```cIMS``` ```DTIMS``` ```FAIMS``` ```SLIM``` ```TIMS``` ```TWIMS``` | False | -| desorption_solvent | Allowable Value | Solvent composition for conducting nanospray desorption electrospray ionization (nanoDESI) or desorption electrospray ionization (DESI). | ```Acetonitrile:Dimethylformamide (ACN:DMF)``` ```Acetonitrile:Water (ACN:H2O)``` ```Ethanol:Dimethylformamide (EtOH:DMF)``` ```Ethanol:Water (EtOH:H2O)``` ```Methanol:Ethanol (MeOH:EtOH)``` ```Methanol:Water (MeOH:H2O)``` | True | -| desorption_solvent_flow_rate_value | Numeric | The rate of flow of the solvent into a spray. | | True | -| desorption_solvent_flow_rate_unit | Allowable Value | Units of the rate of solvent flow. | ```nL/min``` ```uL/min``` | True | -| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | -| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
+--- +layout: page-triary +--- + +# DESI Metadata Attributes + +Fields that are collected for DESI data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | +| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | +| ms_ionization_technique *| | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20``` | +| ms_scan_mode *| | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | +| mass_analysis_polarity *| | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | +| mass_to_charge_range_low_value | | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_to_charge_range_high_value | | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_resolving_power *| | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | +| mass_to_charge_resolving_power | | The peak (m/z) used to calculate the resolving power. | | +| ion_mobility | | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```TIMS``` ```SLIM``` ```FAIMS``` ```DTIMS``` ```cIMS``` ```TWIMS``` | +| matrix_deposition_method | | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_matrix | | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | +| desorption_solvent *| | Solvent composition for conducting nanospray desorption electrospray ionization (nanoDESI) or desorption electrospray ionization (DESI). | ```Acetonitrile:Dimethylformamide (ACN:DMF)``` ```Acetonitrile:Water (ACN:H2O)``` ```Ethanol:Dimethylformamide (EtOH:DMF)``` ```Ethanol:Water (EtOH:H2O)``` ```Methanol:Ethanol (MeOH:EtOH)``` ```Methanol:Water (MeOH:H2O)``` | +| desorption_solvent_flow_rate_value *| | The rate of flow of the solvent into a spray. | | +| desorption_solvent_flow_rate_unit *| | Units of the rate of solvent flow. | ```nL/min``` ```uL/min``` | +| analysis_protocol_doi *| | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| description | | Free-text description of this assay. | | +| overall_protocols_io_doi | | DOI for protocols.io referring to the overall protocol for the assay. | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/DNA-Methylation.md b/docs/assays/metadata/DNA-Methylation.md deleted file mode 100644 index f87faab0..00000000 --- a/docs/assays/metadata/DNA-Methylation.md +++ /dev/null @@ -1,28 +0,0 @@ ---- -layout: page ---- -# DNA Methylation - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial ver0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```DNA Methylation```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: d70bfe24-e82a-46cb-9369-28ae03660d97 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/EnhancedSRS.md b/docs/assays/metadata/EnhancedSRS.md deleted file mode 100644 index 7a906777..00000000 --- a/docs/assays/metadata/EnhancedSRS.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# Enhanced-SRS - -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
diff --git a/docs/assays/metadata/FACS.md b/docs/assays/metadata/FACS.md deleted file mode 100644 index 8c2d8d98..00000000 --- a/docs/assays/metadata/FACS.md +++ /dev/null @@ -1,39 +0,0 @@ ---- -layout: page ---- -# FACS - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| is_erythrocyte_lysis_performed | Radio | Process in which red blood cells (RBCs) are broken down in the sample prior to analysis, thereby allowing researchers to focus primarily on white blood cells (WBCs). | ```Yes,No``` | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | False | -| antibody_reagent_kit | Assigned Value | The kit containing the set of antibodies pre-conjugated with different heavy metal isotopes used to simultaneously detect and quantify multiple protein markers on individual cells by attaching these metal-labeled antibodies to specific cellular targets, essentially acting as the key component for labeling cells with the various markers needed for analysis on the CyTOF machine. | ```Standard BioTools; Maxpar Nuclear Antigen Staining Kit; PN 201603```, ```Standard BioTools; Maxpar Phosphoprotein Staining Kit; PN 201604```, ```Standard BioTools; Maxpar Cell Surface Staining Kit; PN 201601```, ```Standard BioTools; Maxpar Cytoplasmic/Secreted Antigen Staining Kit; PN 201602```, ```Custom``` | True | -| viability_reagent_kit | Assigned Value | The kit used to differentiate between live and dead cells within a sample by selectively staining dead cells with a dye that can be detected by the instrument, allowing researchers to exclude dead cell data from their analysis and ensure accurate results when studying cell populations. | ```Sigma Aldrich; Cisplatin 25mg; PN P4394```, ```Standard BioTools; Cell-ID Cisplatin-198Pt 100 uL; PN 201198```, ```None```, ```Standard BioTools; Cell-ID Intercalator-103Rh 2,000 um; PN 201103B```, ```Standard BioTools; Cell-ID Cisplatin-196Pt 100 uL; PN 201196```, ```Standard BioTools; Cell-ID Cisplatin 100 uL; PN 201064```, ```Standard BioTools; Cell-ID Intercalator-103Rh 500 um; PN 201103A```, ```Standard BioTools; Cell-ID Cisplatin-194Pt 100 uL; PN 201194```, ```Standard BioTools; Cell-ID Cisplatin-195Pt 100 uL; PN 201195```, ```Custom``` | True | -| is_cell_activation_performed | Radio | Process by which ligand is binded to its receptors on a cell, which enhances the cell's ability to respond to various stimuli. | ```Yes,No``` | True | -| activation_stimulus | Textfield | Specific type of stimulus used to provoke cell activation. Examples would include PMA/ionomycin or CD28in/brefeldin A. This field is required if "is_cells_activation performed" is Yes. | | False | -| is_fcr_blocking_applied | Radio | Process by which a reagent has been added to the staining procedure to block the binding of antibodies to Fc receptors (FcRs) on cells, preventing non-specific binding and ensuring that only the intended target antigen is detected by the antibodies; essentially, it helps to minimize false positive signals by preventing antibodies from attaching to the cell via their Fc region instead of the antigen-specific binding site. | ```Yes,No``` | True | -| is_heparin_used | Radio | Indicates whether heparin was used ("Yes") or not ("No") during staining to prevent non-specific binding of metal-labeled antibodies to eosinophils to reduce background noise. | ```Yes,No``` | True | -| loaded_cell_concentration_value | Numeric | The number of cells present within a given volume of liquid for the experiment immediately prior to the experiment, essentially indicating how densely packed the cells are in a solution. | | False | -| loaded_cell_concentration_unit | Assigned Value | Unit of measure for cell concentration, e.g. cells per milliliter (cells/mL). | ```cells/mL``` | False | -| instrument_calibration_bead_kit | Assigned Value | A set of beads of known mass intensity used to adjust the settings of a flow cytometer to ensure accurate measurements. | ```Standard BioTools; EQ Six Element Calibration Beads 100 mL; PN 201245```, ```Standard BioTools; EQ Four Element Calibration Beads 100 mL; PN 201078```, ```None```, ```Standard BioTools; CyTOF Calibration Beads; PN 201073```, ```Custom``` | True | -| calibration_kit_lot_number | Textfield | Manufacturer's lot number for the calibration bead kit used for the experiment. | | True | - -
diff --git a/docs/assays/metadata/GeoMx.md b/docs/assays/metadata/GeoMx.md deleted file mode 100644 index 0747a560..00000000 --- a/docs/assays/metadata/GeoMx.md +++ /dev/null @@ -1,97 +0,0 @@ ---- -layout: page ---- -# GeoMx - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
NGS Version 2 (current) - -## NGS Version 2 (current) - -| attribute | type | description | value | required | -|-----------------------------------------------------|----------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------|------------| -| dataset_type | Textfield | The specific type of dataset being produced. | | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Textfield | The time duration unit of measurement | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Textfield | The time unit of measurement | | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ['Yes', 'No'] | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Textfield | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | -| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | True | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | True | -| target_retrieval_incubation_time_unit | Textfield | The units for target retrieval incubation time value. | | True | -| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | False | -| proteinasek_incubation_time_unit | Textfield | The units for proteinaseK incubation time value. | | False | -| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | True | -| is_roi_segmentation_performed | Allowable Value | Was the image segmented. For GeoMx this refers to whether segmentation was used to split ROIs (regions of interest) into AOIs (areas of interest). | ['Yes', 'No'] | True | -| roi_segmentation_strategy | Textfield | The method of segmentation that was applied in a GeoMx assay. If an overlay was used the overlay image needs to be included in the dataset upload. | | False | -| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | -| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | -| targeted_entity_label | Textfield | State what cell type(s) or functional tissue unit was targeted in this ROI/AOI. | | True | -| targeted_entity_id | Textfield | The ontology ID for the targeted entity. | | False | -| segment_id | Textfield | This is the ID for the area of interest (AOI) in a GeoMx dataset. From "Initial Dataset" spreadsheet (download from within Data Analysis Suite), e.g. 9a828e39-43d8-4051-9bcc-581a520a85d4. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ['Yes', 'No'] | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | True | - -
- -
nCounter Version 2 (current) - -## nCounter Version 2 (current) - -| attribute | type | description | value | required | -|-----------------------------------------------------|----------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------|------------| -| dataset_type | Textfield | The specific type of dataset being produced. | | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Textfield | The time duration unit of measurement | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Textfield | The time unit of measurement | | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ['Yes', 'No'] | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Textfield | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | -| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | True | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | True | -| target_retrieval_incubation_time_unit | Textfield | The units for target retrieval incubation time value. | | True | -| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | False | -| proteinasek_incubation_time_unit | Textfield | The units for proteinaseK incubation time value. | | False | -| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | True | -| is_roi_segmentation_performed | Allowable Value | Was the image segmented. For GeoMx this refers to whether segmentation was used to split ROIs (regions of interest) into AOIs (areas of interest). | ['Yes', 'No'] | True | -| roi_segmentation_strategy | Textfield | The method of segmentation that was applied in a GeoMx assay. If an overlay was used the overlay image needs to be included in the dataset upload. | | False | -| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | -| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | -| targeted_entity_label | Textfield | State what cell type(s) or functional tissue unit was targeted in this ROI/AOI. | | True | -| targeted_entity_id | Textfield | The ontology ID for the targeted entity. | | False | -| segment_id | Textfield | This is the ID for the area of interest (AOI) in a GeoMx dataset. From "Initial Dataset" spreadsheet (download from within Data Analysis Suite), e.g. 9a828e39-43d8-4051-9bcc-581a520a85d4. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ['Yes', 'No'] | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| hybcode_pack_lot_number | Textfield | Enter the lot number noted within the LabWorksheet.txt file (and used in downstream nCounter processing). | | True | -| probe_hybridization_time_value | Numeric | How many hours were the oligo-conjugated RNA or oligo-conjugated antibody probes hybridized with the sample? | | True | -| probe_hybridization_time_unit | Textfield | The units for probe hybridization time value. | | True | -| oligo_probe_panel | Textfield | This is the probe panel used to target genes and/or proteins. In cases where there is a core panel and add-on modules, the core panel should be selected here. If additional panels are used, then they must be included in the "additional_panels_used.csv" file that's uploaded with the dataset. | | True | -| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ['Yes', 'No'] | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | True | - -
diff --git a/docs/assays/metadata/HiFi.md b/docs/assays/metadata/HiFi.md deleted file mode 100644 index ca89eb5c..00000000 --- a/docs/assays/metadata/HiFi.md +++ /dev/null @@ -1,47 +0,0 @@ ---- -layout: page ---- -# HiFi - -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | True | -| spot_size_value | Numeric | FModified progressive staining, Not applicable, Progressive staining, Regressive stainingor assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Allowable Value | The unit for spot size value. | ```um^2``` ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Allowable Value | Units corresponding to inter-spot distance | ```um``` | True | -| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be ```A1, B1, C1, D1```. For HiFi this would be the lane on the flowcell. | ```A1``` ```B1``` ```C1``` ```D1``` ```Lane 1``` ```Lane 2``` ```Lane 3``` ```Lane 4``` ```Lane 5``` ```Lane 6``` ```Lane 7``` ```Lane 8``` | False | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | True | -| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | True | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | True | -| target_retrieval_incubation_time_unit | Allowable Value | The units for target retrieval incubation time value. | ```minute``` | True | -| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | True | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | True | -| proteinasek_incubation_time_unit | Allowable Value | The units for proteinaseK incubation time value. | ```minute``` | True | -| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | -| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | False | - -
diff --git a/docs/assays/metadata/Histology.md b/docs/assays/metadata/Histology.md index 7170d0bb..b02db691 100644 --- a/docs/assays/metadata/Histology.md +++ b/docs/assays/metadata/Histology.md @@ -1,41 +1,68 @@ ---- -layout: page ---- -# Histology - -
Version 2 (Latest) - -## Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| is_batch_staining_done | Allowable Value | Are the slides stained using a linear batch method or individually? | ```Yes``` ```No``` | True | -| is_staining_automated | Allowable Value | Is the slide staining automated with an instrument? | ```Yes``` ```No``` | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| stain_name | Allowable Value | The name of the chemical stains (dyes) applied to histology samples to highlight important features of the tissue as well as to enhance the tissue contrast. | ```AB-PAS``` ```H&E``` ```H-DAB``` ```LFB``` ```PAS``` ```Trichrome ```| True | -| stain_technique | Allowable Value | There are typically three types of stains: progressive, modified progressive, and regressive. Progressive staining occurs when the hematoxylin is added to the tissue without being followed by a differentiator to remove excess dye. With regressive and modified progressive staining, a differentiator is used. | ```Modified progressive staining``` ```Not applicable``` ```Progressive staining``` ```Regressive staining``` | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | - -
+--- +layout: page-triary +--- + +# Histology Metadata Attributes + +Fields that are collected for Histology data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | +| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | +| is_image_preprocessing_required | | Indicates whether image preprocessing is necessary based on the type of acquisition instrument used, such as a microscope or slide scanner. This may involve steps like fusing image tiles to assemble the complete image. Example: Yes | | +| stain_name *| | The name of the chemical stains (dyes) applied to histology samples to highlight important features of the tissue as well as to enhance the tissue contrast. | ```AB-PAS``` ```H&E``` ```H-DAB``` ```LFB``` ```PAS``` ```Trichrome``` | +| stain_technique | | There are typically three types of stains: progressive, modified progressive, and regressive. Progressive staining occurs when the hematoxylin is added to the tissue without being followed by a differentiator to remove excess dye. With regressive and modified progressive staining, a differentiator is used. | ```Modified progressive staining``` ```Not applicable``` ```Progressive staining``` ```Regressive staining``` | +| is_batch_staining_done *| | Are the slides stained using a linear batch method or individually? | | +| is_staining_automated *| | Is the slide staining automated with an instrument? | | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| slide_id | | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| tile_configuration | | The configuration of tiles used for stitching in the assay process. If no tile configuration is applicable, enter "Not applicable". Example: Row-by-row | ```Column-by-column``` ```Not applicable``` ```Snake-by-columns``` ```Row-by-row``` ```Snake-by-rows``` | +| scan_direction | | The direction of imaging, which is necessary for the stitching process. Example: Left-and-down | ```Left-and-down``` ```Right-and-down``` ```Not applicable``` ```Right-and-up``` ```Left-and-up``` | +| tiled_image_columns | | The number of columns used in the stitching process of a tiled image, often referred to as the grid size in the x-dimension. Example: 5 | | +| tiled_image_count | | The total number of raw tiled images captured, which are intended to be stitched together. Example: 75 | | +| intended_tile_overlap_percentage | | The intended percentage of overlap between tiled images. This value serves as the set point, although slight variations may occur during image acquisition due to stage registration. Example: 5 | | +| non_global_files | | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| overall_protocols_io_doi | | DOI for protocols.io referring to the overall protocol for the assay. | | +| principal_investigator | | | | +| pi_email | | Email address for the principal investigator. | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| resolution_z_unit | | The unit of incremental distance between image slices. | ```mm``` ```um``` ```nm``` | +| resolution_z_value | | Optional if assay does not have multiple z-levels. Note that thisis resolution within a given sample: z-pitch (resolution_z_value) is the incrementdistance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stageis moved up or down in increments of 1.5um to capture images of several focalplanes. The best one will be used & the rest discarded. The thickness of the sampleitself is sample metadata. | | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/testing/IMC-2D.md b/docs/assays/metadata/IMC-2D.md similarity index 100% rename from docs/assays/metadata/testing/IMC-2D.md rename to docs/assays/metadata/IMC-2D.md diff --git a/docs/assays/metadata/IMC.md b/docs/assays/metadata/IMC.md deleted file mode 100644 index d4383200..00000000 --- a/docs/assays/metadata/IMC.md +++ /dev/null @@ -1,234 +0,0 @@ ---- -layout: page ---- -# IMC-2D - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
2D IMC Version 2 (Latest) - -## 2D IMC Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| total_run_time_value | Numeric | How long the tissue was on the acquisition instrument. | | True | -| total_run_time_unit | Allowable Value | The units for the total run time unit field. | ```Hour``` ```Minute``` | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes. | | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ```Hz``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - - -
- -
2D IMC Version 1 - -## 2D IMC Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Imaging Mass Cytometry'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| number_of_channels | Numeric | Number of mass channels measured | | True | -| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | -| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | -| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| dual_count_start | Numeric | Threshold for dual counting. | | True | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
2D IMC Version 0 - -## 2D IMC Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Imaging Mass Cytometry'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| number_of_channels | Numeric | Number of mass channels measured | | True | -| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | -| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | -| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| dual_count_start | Numeric | Threshold for dual counting. | | True | -| end_datetime | Datetime | Time stamp indicating end of ablation for ROI | | True | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| start_datetime | Datetime | Time stamp indicating start of ablation for ROI | | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
3D IMC Version 1 (no longer accepting data) - -## 3D IMC Version 1 (no longer accepting data) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['3D Imaging Mass Cytometry'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| number_of_channels | Numeric | Number of mass channels measured | | True | -| number_of_sections | Numeric | Number of sections | | True | -| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | -| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | -| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
3D IMC Version 0 - -## 3D IMC Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['3D Imaging Mass Cytometry'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| number_of_channels | Numeric | Number of mass channels measured | | True | -| number_of_sections | Numeric | Number of sections | | True | -| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | -| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | -| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/Illumina-Spatial.md b/docs/assays/metadata/Illumina-Spatial.md deleted file mode 100644 index 498c6115..00000000 --- a/docs/assays/metadata/Illumina-Spatial.md +++ /dev/null @@ -1,41 +0,0 @@ ---- -layout: page ---- -# Illumina Spatial ver0 - -
Version 0 (current) - -## Version 0 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial v0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | True | -| capture_area_id | Radio | The capture area on the slide that was used during the process. For example, in the case for Visium, this would correspond to areas such as [A1, B1, C1, D1], while for HiFi, it would refer to the lane on the flowcell. Example: A1 | ```A1```, ```B1```, ```C1```, ```D1```, ```Lane 1```, ```Lane 2```, ```Lane 3```, ```Lane 4```, ```Lane 5```, ```Lane 6```, ```Lane 7```, ```Lane 8``` | False | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| preparation_instrument_vendor | Assigned Value | The company that manufactures the instrument used to prepare the sample (e.g., for staining or other processing steps) prior to the assay. If the instrument was custom-built or developed internally, enter "In-House". If no sample preparation occurred, enter "Not applicable". Example: 10X Genomics | ```Thermo Fisher Scientific```, ```SunChrom```, ```Akoya Biosciences```, ```Leica Biosystems```, ```Ionpath```, ```Roche Diagnostics```, ```In-House```, ```Not applicable```, ```Hamamatsu```, ```HTX Technologies```, ```10x Genomics``` | False | -| preparation_instrument_model | Assigned Value | The specific model of the instrument used for sample preparation, such as staining. Manufacturers may offer multiple models with varying features or sensitivities, which can influence how the sample is processed and how the resulting data is interpreted. If no sample preparation occurred, enter "Not applicable". Example: Chromium X | ```AutoStainer XL```, ```ST5020 Multistainer```, ```Visium CytAssist```, ```SunCollect Sprayer```, ```Chromium X```, ```Chromium iX```, ```EVOS M7000```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```Discovery Ultra```, ```Sublimator```, ```Not applicable```, ```TM-Sprayer```, ```M5 Sprayer```, ```M3+ Sprayer```, ```Chromium Controller```, ```Chromium Connect```, ```Custom``` | False | -| capture_area_width_value | Numeric | The width of RNA capture area. Example: 10 | | True | -| capture_area_width_unit | Assigned Value | The unit of measurement for the capture area width value. If the width value is not specified, this field may be left blank. Example: mm | ```mm``` | True | -| capture_area_height_value | Numeric | The height of RNA capture area. Example: 10 | | True | -| capture_area_height_unit | Assigned Value | The unit of measurement for the capture area height value. If the height value is not specified, this field may be left blank. Example: mm | ```mm``` | True | -| spatial_discreatization_method | Assigned Value | The segmentation method used to divide the capture are into smaller, defined regions for analysis. Example: Cell segmentation | ```Square binning```, ```Cell segmentation```, ```Hexagonal binning``` | True | -| bin_size | Textfield | The size (in µm) of each discrete spatial unit ("bin") used to partition the capture area in bin-based spatial discretization. Example: 100 | | False | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/LC-MS.md b/docs/assays/metadata/LC-MS.md index 75cdce06..a98bf37b 100644 --- a/docs/assays/metadata/LC-MS.md +++ b/docs/assays/metadata/LC-MS.md @@ -1,296 +1,89 @@ ---- -layout: page ---- -# LC-MS - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 4 (Latest) - -## Version 4 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | -| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | False | -| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | -| ion_mobility | Allowable Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```cIMS``` ```DTIMS``` ```FAIMS``` ```SLIM``` ```TIMS``` ```TWIMS``` | False | -| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | -| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | -| label_name | Textfield | If the samples were labeled (e.g. TMT), provide the name/ID of the label on this sample. Leave blank if not applicable. | | False | -| lc_instrument_vendor | Allowable Value | The manufacturer of the instrument used for liquid chromatography. | ```Agilent Technologies``` ```Bruker``` ```Evosep``` ```In-House``` ```Sciex``` ```Thermo Fisher Scientific``` ```Waters``` | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for liquid chromatography. | | False | -| lc_column_model | Textfield | The model number/name of the liquid chromatography column. If it is a custom self-packed, pulled tip capillary is used enter “Pulled tip capilary”. | | False | -| lc_resin | Textfield | Details of the resin used for liquid chromatography, including vendor, particle size, pore size. | | False | -| lc_column_length_value | Numeric | Liquid chromatography column length. | | False | -| lc_column_length_unit | Allowable Value | Units for liquid chromatography column length (typically cm). | ```um``` ```mm``` ```cm``` | False | -| lc_temperature_value | Numeric | Liquid chromatography temperature. | | False | -| lc_inner_diameter_value | Numeric | Liquid chromatography column inner diameter. | | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_gradient_value | Numeric | Liquid chromatography gradient. | | False | -| lc_gradient_unit | Allowable Value | Unit for liquid chromatography gradient | ```Minute``` | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A. | | False | -| lc_mobile_phase_b | Textfield | | | False | -| spatial_sampling_technique | Allowable Value | | ```LCM``` ```LESA``` ```microLESA``` ```microPOTS``` ```nanoPOTS``` ```nanoSPLITS``` | False | -| spatial_sampling_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targeted in the spatial profiling experiment. Leave blank if data are generated in imaging mode without a specific target structure. | | False | -| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependent acquisition), DIA (Data-independent acquisition), SRM (multiple reaction monitoring), or PRM (parallel reaction monitoring). | ```DDA``` ```PRM``` ```DIA``` ```SRM``` | False | -| lc_column_vendor | Allowable Value | The manufacturer of the liquid chromatography column unless self-packed, pulled tip capillary is used. | ```Bruker``` ```Evosep``` ```In-House``` ```IonOpticks``` ```Thermo Fisher Scientific``` ```Waters``` | False | -| lc_temperature_unit | Allowable Value | | ```Celsius``` | False | -| lc_inner_diameter_unit | Allowable Value | | ```um``` ```mm``` ```cm``` | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ```mL/min``` ```nL/min``` | False | -| spatial_sampling_type | Allowable Value | Specifies whether or not the analysis was performed in a spatially targeted manner. Spatial profiling experiments target specific tissue foci but do not necessarily generate images. Spatial imaging expriments collect data from a regular array (pixels) that can be visualized as heat maps of ion intensity at each location (molecular images). Leave blank if data are derived from bulk analysis. | ```Imaging``` ```Profiling``` | False | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
Version 3 - -## Version 3 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['3'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | -| assay_type | Allowable Value | Bottom-up refers to analyzing proteins in a sample by digesting themto peptides. Top-down refers to analyzing whole proteins without digestion. LC-MSand MS are for lipids/metabolites. LC-MS Bottom-Up and MS Bottom-Up are for peptides.LC-MS Top-Down and MS Top-Down are for proteins. | ['LC-MS', 'MS', 'LC-MS Bottom-Up', 'MS Bottom-Up', 'LC-MS Top-Down', 'MS Top-Down'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids', 'peptides', 'phosphopeptides', 'glycans'] | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| dms | Allowable Value | Was differential mobility spectrometry used in this assay? | ['Yes','No'] | True | -| ms_source | Allowable Value | The ion source type used for surface sampling. | ['ESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| mass_resolving_power | Numeric | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | False | -| mz_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | -| ion_mobility | Allowable Value | Specifies whether or not ion mobility spectrometry was performed andwhich technology was used. Technologies for measuring ion mobility: TravelingWave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS),High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube IonMobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ['TIMS', 'TWIMS', 'FAIMS', 'DTIMS', 'SLIMS'] | False | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | -| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | -| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | -| label_name | Textfield | If the samples were labeled (e.g. TMT), provide the name/ID of thelabel on this sample. | | False | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | -| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | -| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | -| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | -| lc_length_value | Numeric | LC column length | | False | -| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | -| lc_temp_value | Numeric | LC temperature | | False | -| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | -| lc_id_value | Numeric | LC column inner diameter (microns) | | False | -| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | -| lc_gradient | Textfield | LC gradient | | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | -| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | -| spatial_type | Allowable Value | Specifies whether or not the analysis was performed in a spatialy targetedmanner and the technique used for spatial sampling. For example, Laser-capturemicrodissection (LCM), Liquid Extraction Surface Analysis (LESA), NanodropletProcessing in One pot for Trace Samples (nanoPOTS). | ['LCM', 'LESA', 'nanoPOTS', 'microLESA'] | False | -| spatial_sampling_type | Allowable Value | Specifies whether or not the analysis was performed in a spatiallytargeted manner. Spatial profiling experiments target specific tissue foci butdo not necessarily generate images. Spatial imaging expriments collect data froma regular array (pixels) that can be visualized as heat maps of ion intensityat each location (molecular images). Leave blank if data are derived from bulkanalysis. | ['profiling', 'imaging'] | False | -| spatial_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targetedin the spatial profiling experiment. Leave blank if data are generated in imagingmode without a specific target structure. | | False | -| resolution_x_value | Numeric | The width of a pixel. | | False | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | False | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 2 - -## Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | -| assay_type | Allowable Value | Bottom-up refers to analyzing proteins in a sample by digesting themto peptides. Top-down refers to analyzing whole proteins without digestion. LC-MSand MS are for lipids/metabolites. LC-MS Bottom-Up and MS Bottom-Up are for peptides.LC-MS Top-Down and MS Top-Down are for proteins. | ['LC-MS', 'MS', 'LC-MS Bottom-Up', 'MS Bottom-Up', 'LC-MS Top-Down', 'MS Top-Down'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids', 'peptides', 'phosphopeptides', 'glycans'] | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| ms_source | Allowable Value | The ion source type used for surface sampling. | ['ESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| mass_resolving_power | Numeric | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | False | -| mz_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | -| ion_mobility | Allowable Value | Specifies whether or not ion mobility spectrometry was performed andwhich technology was used. Technologies for measuring ion mobility: TravelingWave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS),High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube IonMobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ['TIMS', 'TWIMS', 'FAIMS', 'DTIMS', 'SLIMS'] | False | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | -| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | -| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | -| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | -| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | -| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | -| lc_length_value | Numeric | LC column length | | False | -| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | -| lc_temp_value | Numeric | LC temperature | | False | -| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | -| lc_id_value | Numeric | LC column inner diameter (microns) | | False | -| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | -| lc_gradient | Textfield | LC gradient | | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | -| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | -| spatial_type | Allowable Value | Specifies whether or not the analysis was performed in a spatialy targetedmanner and the technique used for spatial sampling. For example, Laser-capturemicrodissection (LCM), Liquid Extraction Surface Analysis (LESA), NanodropletProcessing in One pot for Trace Samples (nanoPOTS). | ['LCM', 'LESA', 'nanoPOTS', 'microLESA'] | False | -| spatial_sampling_type | Allowable Value | Specifies whether or not the analysis was performed in a spatiallytargeted manner. Spatial profiling experiments target specific tissue foci butdo not necessarily generate images. Spatial imaging expriments collect data froma regular array (pixels) that can be visualized as heat maps of ion intensityat each location (molecular images). Leave blank if data are derived from bulkanalysis. | ['profiling', 'imaging'] | False | -| spatial_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targetedin the spatial profiling experiment. Leave blank if data are generated in imagingmode without a specific target structure. | | False | -| resolution_x_value | Numeric | The width of a pixel. | | False | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | False | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 1 - -## Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['LC-MS (metabolomics)', 'LC-MS/MS (label-free proteomics)', 'MS (shotgun lipidomics)'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| ms_source | Textfield | The ion source type used for surface sampling (MALDI, MALDI-2, DESI,or SIMS) or LC-MS/MS data acquisition (nESI) | | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | -| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | -| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | -| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | -| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | -| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | -| lc_length_value | Numeric | LC column length | | False | -| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | -| lc_temp_value | Numeric | LC temperature | | False | -| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | -| lc_id_value | Numeric | LC column inner diameter (microns) | | False | -| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | -| lc_gradient | Textfield | LC gradient | | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | -| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | -| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['LC-MS (metabolomics)', 'LC-MS/MS (label-free proteomics)', 'MS (shotgun lipidomics)'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| ms_source | Textfield | The ion source type used for surface sampling (MALDI, MALDI-2, DESI,or SIMS) or LC-MS/MS data acquisition (nESI) | | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | -| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | -| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | -| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | -| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | -| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | -| lc_length_value | Numeric | LC column length | | False | -| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | -| lc_temp_value | Numeric | LC temperature | | False | -| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | -| lc_id_value | Numeric | LC column inner diameter (microns) | | False | -| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | -| lc_gradient | Textfield | LC gradient | | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | -| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | -| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
\ No newline at end of file +--- +layout: page-triary +--- + +# LC-MS Metadata Attributes + +Fields that are collected for LC-MS data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| ms_ionization_technique *| | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20``` | +| ms_scan_mode *| | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 for TMT) | ```MS1``` ```MS2``` ```MS3``` | +| mass_analysis_polarity *| | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | +| mass_to_charge_range_low_value | | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_to_charge_range_high_value | | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_resolving_power | | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | +| mass_to_charge_resolving_power | | The peak (m/z) used to calculate the resolving power. | | +| ion_mobility | | Specifies whether or not ion mobility spectrometry was performed and which technology was used. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ```TIMS``` ```SLIM``` ```FAIMS``` ```DTIMS``` ```cIMS``` ```TWIMS``` | +| data_collection_mode *| | Mode of data collection in tandem MS assays. Either DDA (Data-dependent acquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring), or PRM (parallel reaction monitoring). | ```DDA``` ```PRM``` ```DIA``` ```SRM``` | +| label_name | | If the samples were labeled (e.g. TMT), provide the name/ID of the label on this sample. | | +| lc_instrument_vendor | | The manufacturer of the instrument used for LC | ```Thermo Fisher Scientific``` ```Sciex``` ```In-House``` ```Agilent Technologies``` ```Waters``` ```Bruker``` ```Evosep``` | +| lc_instrument_model | | The model number/name of the instrument used for LC | | +| lc_column_vendor | | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulled tip capilary is used | ```Thermo Fisher Scientific``` ```In-House``` ```Waters``` ```Bruker``` ```Evosep``` ```IonOpticks``` | +| lc_column_model | | The model number/name of the LC Column - IF custom self-packed, pulled tip calillary is used enter "Pulled tip capilary" | | +| lc_resin | | Details of the resin used for lc, including vendor, particle size, pore size | | +| lc_column_length_value | | Liquid chromatography column length. | | +| lc_column_length_unit | | Units for liquid chromatography column length (typically cm). | ```um``` ```mm``` ```cm``` | +| lc_temperature_value | | Liquid chromatography temperature. | | +| lc_temperature_unit | | | ```celsius``` | +| lc_inner_diameter_value | | Liquid chromatography column inner diameter. | | +| lc_inner_diameter_unit | | | ```um``` ```mm``` ```cm``` | +| lc_flow_rate_value | | Value of flow rate. | | +| lc_flow_rate_unit | | Units of flow rate. | ```nL/min``` ```mL/min``` | +| lc_gradient_value | | Liquid chromatography gradient. | | +| lc_gradient_unit | | Unit for liquid chromatography gradient | ```minute``` | +| lc_mobile_phase_a | | Composition of mobile phase A | | +| lc_mobile_phase_b | | Composition of mobile phase B | | +| spatial_sampling_technique | | | ```nanoSPLITS``` ```nanoPOTS``` ```LESA``` ```microPOTS``` ```LCM``` ```microLESA``` | +| spatial_sampling_target | | Specifies the cell-type or functional tissue unit (FTU) that is targeted in the spatial profiling experiment. Leave blank if data are generated in imaging mode without a specific target structure. | | +| spatial_sampling_type | | Specifies whether or not the analysis was performed in a spatially targeted manner. Spatial profiling experiments target specific tissue foci but do not necessarily generate images. Spatial imaging expriments collect data from a regular array (pixels) that can be visualized as heat maps of ion intensity at each location (molecular images). Leave blank if data are derived from bulk analysis. | ```Imaging``` ```Profiling``` | +| analysis_protocol_doi *| | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | +| acquisition_protocol_doi | | | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| description | | Free-text description of this assay. | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| protocols_io_doi | | DOI for protocols.io referring to the protocol for this assay. | | +| overall_protocols_io_doi | | DOI for protocols.io for the overall process for this assay. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| processing_search | | Software for analyzing and searching LC-MS/MS omics data | | +| labeling | | Indicates whether samples were labeled prior to MS analysis (e.g., TMT) | | +| dms | | Was differential mobility spectrometry used in this assay? | | +| resolution_x_unit | | The unit of measurement of the width of a pixel. | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. | | +| resolution_y_unit | | The unit of measurement of the height of a pixel. | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel | | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/testing/Light-Sheet.md b/docs/assays/metadata/Light-Sheet.md similarity index 100% rename from docs/assays/metadata/testing/Light-Sheet.md rename to docs/assays/metadata/Light-Sheet.md diff --git a/docs/assays/metadata/LightSheet.md b/docs/assays/metadata/LightSheet.md deleted file mode 100644 index 4eb8950d..00000000 --- a/docs/assays/metadata/LightSheet.md +++ /dev/null @@ -1,146 +0,0 @@ ---- -layout: page ---- -# Light-Sheet - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 3 (latest) - -## Version 3 (latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
Version 2 - -## Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Light Sheet'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| range_z_value | Numeric | The total range of the z axis. | | True | -| range_z_unit | Allowable Value | The unit of range_z_value. | ['nm', 'um'] | False | -| step_z_value | Numeric | The number of optical sections in z axis range. | | True | -| increment_z_value | Numeric | The distance between sequential optical sections. | | True | -| increment_z_unit | Allowable Value | The units of increment z value. | ['nm', 'um'] | False | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 1 - -## Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Light Sheet'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| resolution_z_value | Numeric | The distance at which two objects along the detection z-axis can bedistinguished (resolved as 2 objects). | | True | -| resolution_z_unit | Allowable Value | The unit of distance at which two objects along the detection z-axiscan be distinguished (resolved as 2 objects). | ['mm', 'um', 'nm'] | False | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Light Sheet'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| resolution_z_value | Numeric | The distance at which two objects along the detection z-axis can bedistinguished (resolved as 2 objects). | | True | -| resolution_z_unit | Allowable Value | The unit of distance at which two objects along the detection z-axiscan be distinguished (resolved as 2 objects). | ['mm', 'um', 'nm'] | False | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/MALDI.md b/docs/assays/metadata/MALDI.md index ccf79e7c..ae760058 100644 --- a/docs/assays/metadata/MALDI.md +++ b/docs/assays/metadata/MALDI.md @@ -1,171 +1,67 @@ ---- -layout: page ---- -# MALDI - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Maldi Version 2 (latest) - -## Maldi Version 2 (latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | True | -| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | -| ion_mobility | Allowable Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```cIMS``` ```DTIMS``` ```FAIMS``` ```SLIM``` ```TIMS``` ```TWIMS``` | False | -| matrix_deposition_method | Allowable Value | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| preparation_matrix | Allowable Value | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | -| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
IMS Version 2 - -## IMS Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['MALDI-IMS', 'SIMS-IMS', 'NanoDESI', 'DESI'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids', 'peptides', 'phosphopeptides', 'glycans'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| ms_source | Allowable Value | The ion source type used for surface sampling (MALDI, MALDI-2, DESI, nanoDESI or SIMS). | ['MALDI', 'MALDI-2', 'LDI', 'LA', 'SIMS-C60', 'SIMS-H2O', 'DESI', 'nanoDESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| mass_resolving_power | Numeric | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | True | -| mz_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | -| ion_mobility | Allowable Value | Specifies whether or not ion mobility spectrometry was performed and which technology was used. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ['TIMS', 'TWIMS', 'FAIMS', 'DTIMS', 'SLIMS'] | False | -| ms_scan_mode | Allowable Value | Scan mode refers to the number of steps in the separation of fragments. | ['MS', 'MS/MS', 'MS3'] | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| preparation_type | Textfield | Common methods of depositing matrix for MALDI imaging include robotic spotting, electrospray deposition, and spray-coating with an airbrush. | | False | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | False | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | False | -| preparation_maldi_matrix | Textfield | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the laser. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | | False | -| desi_solvent | Textfield | Solvent composition for conducting nanospray desorption electrospray ionization (nanoDESI) or desorption electrospray ionization (DESI). | | False | -| desi_solvent_flow_rate | Numeric | The rate of flow of the solvent into a spray. | | False | -| desi_solvent_flow_rate_unit | Allowable Value | Units of the rate of solvent flow. | ['uL/minute'] | False | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
IMS Version 1 - -## IMS Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['MALDI-IMS'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| ms_source | Allowable Value | The ion source type used for surface sampling (MALDI, MALDI-2, DESI, or SIMS) or LC-MS/MS data acquisition (nESI) | ['MALDI', 'MALDI-2', 'DESI', 'SIMS', 'nESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| preparation_type | Textfield | Common methods of depositing matrix for MALDI imaging include robotic spotting, electrospray deposition, and spray-coating with an airbrush. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| preparation_maldi_matrix | Textfield | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the laser. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
IMS Version 0 - -## IMS Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['MALDI-IMS'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| ms_source | Allowable Value | The ion source type used for surface sampling (MALDI, MALDI-2, DESI, or SIMS) or LC-MS/MS data acquisition (nESI) | ['MALDI', 'MALDI-2', 'DESI', 'SIMS', 'nESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| preparation_type | Textfield | Common methods of depositing matrix for MALDI imaging include robotic spotting, electrospray deposition, and spray-coating with an airbrush. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| preparation_maldi_matrix | Textfield | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the laser. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
+--- +layout: page-triary +--- + +# MALDI Metadata Attributes + +Fields that are collected for MALDI data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | +| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | +| ms_ionization_technique *| | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20``` | +| ms_scan_mode *| | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | +| mass_analysis_polarity *| | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | +| mass_to_charge_range_low_value | | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_to_charge_range_high_value | | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_resolving_power *| | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | +| mass_to_charge_resolving_power | | The peak (m/z) used to calculate the resolving power. | | +| ion_mobility | | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```TIMS``` ```SLIM``` ```FAIMS``` ```DTIMS``` ```cIMS``` ```TWIMS``` | +| matrix_deposition_method *| | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | +| preparation_instrument_vendor *| | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model *| | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_matrix *| | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | +| analysis_protocol_doi *| | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| description | | Free-text description of this assay. | | +| section_prep_protocols_io_doi | | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | +| overall_protocols_io_doi | | DOI for protocols.io referring to the overall protocol for the assay. | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/MERFISH.md b/docs/assays/metadata/MERFISH.md deleted file mode 100644 index bb457b7d..00000000 --- a/docs/assays/metadata/MERFISH.md +++ /dev/null @@ -1,47 +0,0 @@ ---- -layout: page ---- -# MERFISH -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | False | -| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | False | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | False | -| target_retrieval_incubation_time_unit | Allowable Value | The units for target retrieval incubation time value. | ```minute``` | False | -| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | False | -| proteinasek_incubation_time_unit | Allowable Value | The units for proteinaseK incubation time value. | ```minute``` | False | -| probe_hybridization_time_value | Numeric | How long was the oligo-conjugated RNA or oligo-conjugated antibody probes hybridized with the sample? | | False | -| probe_hybridization_time_unit | Allowable Value | The units for probe hybridization time value. | ```Hour``` ```Minute``` | False | -| oligo_probe_panel | Allowable Value | This is the probe panel used to target genes and/or proteins. In cases where there is a core panel and add-on modules, the core panel should be selected here. If additional panels are used, then they must be included in the "additional_panels_used.csv" file that's uploaded with the dataset. | ```10x Genomics; Chromium Fixed RNA Kit``` ```Human Transcriptome``` ```4 rxns x 1 BC; PN 1000474``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```16 rxns; PN 1000420``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```64 rxns; PN 1000456``` ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363``` ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365``` ```Custom``` ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-HuWTA-4``` ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-MsWTA-4``` | True | -| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ```Yes``` ```No``` | True | -| number_of_panel_targets | Numeric | How many genes, RNA isoforms or RNA regions are targeted by probes. | | True | -| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | False | -| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | -| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | -| number_of_additional_stains | Numeric | This would be minimally 2 (always include DAPI and polyT) and can include 6 more. | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - -
diff --git a/docs/assays/metadata/MIBI.md b/docs/assays/metadata/MIBI.md index 6c06a9b1..6e500c70 100644 --- a/docs/assays/metadata/MIBI.md +++ b/docs/assays/metadata/MIBI.md @@ -1,104 +1,86 @@ ---- -layout: page ---- -# MIBI - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 2 (latest) - -## Version 2 (latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| roi_description | Textfield | A description of the anatomical structure being captured in the region of interest (ROI). | | True | -| roi_id | Numeric | Multiple images are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The ROI ID is a number from 1 to N representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the "Acquisition ID" and the "ROI ID" indicate the slide-ROI represented in the image. | | True | -| area_normalized_ion_dose_value | Numeric | Number of primary ions delivered to the sample per unit area. | | True | -| area_normalized_ion_dose_unit | Allowable Value | Area normalized ion dose unit. | ```nA*hr/mm2``` | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes. | | True | -| pixel_dwell_time_value | Numeric | Resident time of primary ion beam on each pixel to ionize it. | | True | -| pixel_dwell_time_unit | Allowable Value | Pixel dwell time unit. | ```ms``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - -
- - -
Version 1 - -## Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|--------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['MIBI'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| resolution_x_value | Numeric | The width of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of width of a pixel.(nm) | ['mm', 'um', 'nm'] | False | -| resolution_y_value | Numeric | The height of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of height of a pixel. (nm) | ['mm', 'um', 'nm'] | False | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| area_normalized_ion_dose_unit | Allowable Value | Area normalized ion dose unit | ['nA*hr/mm2'] | False | -| area_normalized_ion_dose_value | Numeric | Number of primary ions delivered to the sample per unit area | | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes | | True | -| dual_count_start | Numeric | Threshold for dual counting. | | True | -| end_datetime | Datetime | Time stamp indicating end of ablation for ROI | | True | -| pixel_dwell_time_value | Numeric | Resident time of primary ion beam on each pixel. | | True | -| pixel_dwell_time_unit | Allowable Value | Pixel dwell time unit. | ['ms'] | False | -| pixel_size_x_value | Numeric | Width value of the pixel or voxel measurement (distinct from the image resolution_x_value). | | True | -| pixel_size_x_unit | Allowable Value | Width unit of the pixel or voxel measurement. | ['nm'] | False | -| pixel_size_y_value | Numeric | Length value of the pixel or voxel measurement (distinct from the image resolution_y_value). | | True | -| pixel_size_y_unit | Allowable Value | Length unit of the pixel or voxel measurement. | ['nm'] | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare the sample for the assay. | ['Custom', 'Ionpath'] | True | -| preparation_instrument_model | Allowable Value | The model number/name of the instrument used to prepare the sample for the assay | ['Custom', 'MIBIscope 1', 'MIBIscope 2'] | True | -| primary_ion | Allowable Value | Primary ion. | ['Xe'] | True | -| primary_ion_current_value | Numeric | Primary ion current value. | | True | -| primary_ion_current_unit | Allowable Value | Primary ion current unit, typically nA or pA | ['nA', 'pA'] | False | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| start_datetime | Datetime | Time stamp indicating start of ablation for ROI | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
+--- +layout: page-triary +--- + +# MIBI Metadata Attributes + +Fields that are collected for MIBI data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| number_of_antibodies *| | Number of antibodies | | +| number_of_channels *| | Number of fluorescent channels imaged during each cycle. | | +| slide_id *| | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| roi_description *| | A description of the region of interest (ROI) captured in the image. | | +| roi_id *| | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | +| acquisition_id *| | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | +| area_normalized_ion_dose_value *| | Number of primary ions delivered to the sample per unit area | | +| area_normalized_ion_dose_unit *| | Area normalized ion dose unit | ```nA*hr/mm2``` | +| data_precision_bytes *| | Numerical data precision in bytes | | +| pixel_dwell_time_value *| | Resident time of primary ion beam on each pixel. | | +| pixel_dwell_time_unit *| | Pixel dwell time unit. | ```ms``` | +| antibodies_path *| | Relative path to file with antibody information for this dataset. | | +| primary_ion *| | Primary ion. | ```Xe``` | +| primary_ion_current_unit | | Primary ion current unit, typically nA or pA | ```nA``` ```pA``` | +| primary_ion_current_value *| | Primary ion current value. | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| description | | Free-text description of this assay. | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| protocol_io_doi | | | | +| reagent_prep_protocols_io_doi | | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | +| preparation_instrument_model | | The model number/name of the instrument used to prepare the sample for the assay | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare the sample for the assay. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| segment_data_format | | This refers to the data type, which is a "float" for the IMC counts. | ```float``` ```integer``` ```string``` | +| signal_type | | Type of signal measured per channel (usually dual counts) | ```dual count``` ```pulse count``` ```intensity value``` | +| dual_count_start | | Threshold for dual counting. | | +| start_datetime | | Time stamp indicating start of ablation for ROI | | +| end_datetime | | Time stamp indicating end of ablation for ROI | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| max_x_width_unit | | Units of image width of the ROI acquisition | ```um``` | +| max_x_width_value | | Image width value of the ROI acquisition | | +| max_y_height_unit | | Units of image height of the ROI acquisition | ```um``` | +| max_y_height_value | | Image height value of the ROI acquisition | | +| pixel_size_x_unit | | Width unit of the pixel or voxel measurement. | ```nm``` | +| pixel_size_x_value | | Width value of the pixel or voxel measurement (distinct from the image resolution_x_value). | | +| pixel_size_y_unit | | Length unit of the pixel or voxel measurement. | ```nm``` | +| pixel_size_y_value | | Length value of the pixel or voxel measurement (distinct from the image resolution_y_value). | | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/MPLEx.md b/docs/assays/metadata/MPLEx.md deleted file mode 100644 index d7c3f0b7..00000000 --- a/docs/assays/metadata/MPLEx.md +++ /dev/null @@ -1,59 +0,0 @@ ---- -layout: page ---- -# MPLEx - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Assigned Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode```, ```Positive ion mode```, ```Negative ion mode``` | True | -| mass_to_charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass_to_charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | False | -| mass_to_charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | -| ion_mobility | Assigned Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```TIMS```, ```SLIM```, ```FAIMS```, ```DTIMS```, ```cIMS```, ```TWIMS``` | False | -| ms_ionization_technique | Assigned Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```MALDI```, ```SIMS-C60```, ```LDI```, ```HESI```, ```nanoDESI```, ```MALDI-2```, ```DESI```, ```LA```, ```SIMS-H20```, ```ESI``` | True | -| ms_scan_mode | Assigned Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS2```, ```MS1```, ```MS3``` | True | -| label_name | Textfield | If the samples were labeled (e.g. TMT), provide the name/ID of the label on this sample. Leave blank if not applicable. | | False | -| lc_instrument_vendor | Assigned Value | The manufacturer of the instrument used for liquid chromatography. | ```Thermo Fisher Scientific```, ```Sciex```, ```In-House```, ```Agilent Technologies```, ```Waters```, ```Bruker```, ```Evosep``` | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for liquid chromatography. | | False | -| lc_column_model | Textfield | The model number/name of the liquid chromatography column. If it is a custom self-packed, pulled tip capillary is used enter “Pulled tip capilary”. | | False | -| lc_resin | Textfield | Details of the resin used for liquid chromatography, including vendor, particle size, pore size. | | False | -| lc_column_length_value | Numeric | Liquid chromatography column length. | | False | -| lc_column_length_unit | Assigned Value | Units for liquid chromatography column length (typically cm). | ```um```, ```mm```, ```cm``` | False | -| lc_temperature_value | Numeric | Liquid chromatography temperature. | | False | -| lc_inner_diameter_value | Numeric | Liquid chromatography column inner diameter. | | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_gradient_value | Numeric | Liquid chromatography gradient. | | False | -| lc_gradient_unit | Assigned Value | Unit for liquid chromatography gradient | ```minute``` | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A. | | False | -| lc_mobile_phase_b | Textfield | | | False | -| spatial_sampling_technique | Assigned Value | | ```nanoSPLITS```, ```nanoPOTS```, ```LESA```, ```microPOTS```, ```LCM```, ```microLESA``` | False | -| spatial_sampling_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targeted in the spatial profiling experiment. Leave blank if data are generated in imaging mode without a specific target structure. | | False | -| analysis_protocol_doi | Link | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| data_collection_mode | Assigned Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependent acquisition), DIA (Data-independent acquisition), SRM (multiple reaction monitoring), or PRM (parallel reaction monitoring). | ```DDA```, ```PRM```, ```DIA```, ```SRM``` | False | -| lc_column_vendor | Assigned Value | The manufacturer of the liquid chromatography column unless self-packed, pulled tip capillary is used. | ```Thermo Fisher Scientific```, ```In-House```, ```Waters```, ```Bruker```, ```Evosep```, ```IonOpticks``` | False | -| lc_temperature_unit | Assigned Value | | ```celsius``` | False | -| lc_inner_diameter_unit | Assigned Value | | ```um```, ```mm```, ```cm``` | False | -| lc_flow_rate_unit | Assigned Value | Units of flow rate. | ```nL/min```, ```mL/min``` | False | -| spatial_sampling_type | Assigned Value | Specifies whether or not the analysis was performed in a spatially targeted manner. Spatial profiling experiments target specific tissue foci but do not necessarily generate images. Spatial imaging expriments collect data from a regular array (pixels) that can be visualized as heat maps of ion intensity at each location (molecular images). Leave blank if data are derived from bulk analysis. | ```Imaging```, ```Profiling``` | False | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/testing/MUSIC-(CEDAR).md b/docs/assays/metadata/MUSIC-(CEDAR).md similarity index 100% rename from docs/assays/metadata/testing/MUSIC-(CEDAR).md rename to docs/assays/metadata/MUSIC-(CEDAR).md diff --git a/docs/assays/metadata/MUSIC.md b/docs/assays/metadata/MUSIC.md index 58c71288..48a650ac 100644 --- a/docs/assays/metadata/MUSIC.md +++ b/docs/assays/metadata/MUSIC.md @@ -1,58 +1,70 @@ ---- -layout: page ---- -# MUSIC - -
Current Metadata Attributes - -## Current Metadata Attributes - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```14-17,14,14``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | False | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | False | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | False | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```Not applicable``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,78``` ```10,48,86``` ```0,20-23,41-44``` ```Not applicable``` | True | -| amount_of_input_analyte_unit | Allowable Value | Units of amount of entity input to assay value | ```ug``` ```ng``` | False | - -
\ No newline at end of file +--- +layout: page-triary +--- + +# MUSIC Metadata Attributes + +Fields that are collected for MUSIC data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | +| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | +| barcode_offset *| | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | +| barcode_read *| | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | +| barcode_size *| | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | +| umi_offset *| | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | +| umi_read *| | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | +| umi_size *| | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | +| assay_input_entity *| | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | +| number_of_input_cells_or_nuclei | | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | +| amount_of_input_analyte_value | | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | +| amount_of_input_analyte_unit | | Units of amount of entity input to assay value | ```ug``` ```ng``` | +| library_adapter_sequence *| | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | +| library_average_fragment_size *| | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | +| library_input_amount_value | | The amount of cDNA, after amplification, that was used for library construction. | | +| library_input_amount_unit | | unit of library input amount value | ```ng``` ```ul``` | +| library_output_amount_value | | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | +| library_output_amount_unit | | Units of library final yield. | ```ng``` ```ul``` | +| library_concentration_value *| | The concentration value of the pooled library samples submitted for sequencing. | | +| library_concentration_unit *| | Unit of library concentration value. | ```ng/ul``` ```nM``` | +| library_layout *| | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | +| library_preparation_kit *| | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```1 slides``` ```4 reactions; PN 1000338``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 reactions; PN 1000187``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | +| sample_indexing_kit *| | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001``` | +| sample_indexing_set | | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | +| is_technical_replicate *| | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | | +| expected_entity_capture_count | | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | +| sequencing_reagent_kit *| | Reagent kit used for sequencing | ```Custom``` ```Illumina``` ```HiSeq 3000/4000 PE Cluster Kit PE-410-1001``` ```PN 1000283, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles)``` ```PN 20046811, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles)``` ```PN 20046812, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles)``` ```PN 20046813, Illumina``` ```NextSeq 2000 P3 Reagent Kit (300 Cycles)``` ```PN 20040561, Illumina``` ```NextSeq 2000 P3 Reagents Kit (100 Cycles)``` ```PN 20040559, Illumina``` ```NextSeq 500/550 Hi Output Kit 150 Cycles``` ```v2.5``` ```PN 20024907, Illumina``` ```NextSeq 500/550 Hi Output Kit 75 Cycles v2.5``` ```PN 20024906, Illumina``` ```NextSeq 500/550 Mid Output Kit 150 Cycles v2.5``` ```PN 20024904, Illumina``` ```NovaSeq 6000 S1 Reagent Kit (200 Cycles)``` ```PN 20012864, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028319, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028318, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles)``` ```PN 20028317, Illumina``` ```NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028316, Illumina``` ```NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles)``` ```PN 20028312, Illumina``` ```NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028313, Illumina``` ```NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles)``` ```PN 20028401, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (100 Cycle)``` ```PN 20104703, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (200 Cycle)``` ```PN 20104704, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (300 Cycle)``` ```PN 20104705, Illumina``` ```NovaSeq X Series 10B Reagent Kit (100 Cycle)``` ```PN 20085596, Illumina``` ```NovaSeq X Series 10B Reagent Kit (200 Cycle)``` ```PN 20085595, Illumina``` ```NovaSeq X Series 10B Reagent Kit (300 Cycle)``` ```PN 20085594``` | +| sequencing_read_format *| | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | +| sequencing_batch_id | | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| capture_batch_id | | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_instrument_kit | | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| diff --git a/docs/assays/metadata/Olink.md b/docs/assays/metadata/Olink.md deleted file mode 100644 index 2a36ed5b..00000000 --- a/docs/assays/metadata/Olink.md +++ /dev/null @@ -1,28 +0,0 @@ ---- -layout: page ---- -# Olink - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/PhenoCycler.md b/docs/assays/metadata/PhenoCycler.md deleted file mode 100644 index 7883ef67..00000000 --- a/docs/assays/metadata/PhenoCycler.md +++ /dev/null @@ -1,36 +0,0 @@ ---- -layout: page ---- -# PhenoCycler - -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| number_of_biomarker_imaging_rounds | Numeric | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo application, 2. fluor application, 3. washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | True | -| number_of_total_imaging_rounds | Numeric | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| total_run_time_value | Numeric | How long the tissue was on the acquisition instrument. | | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| total_run_time_unit | Allowable Value | The units for the total run time unit field. | ```Hour``` ```Minute``` | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| nuclear_marker_or_stain | Allowable Value | For markers, an antibody-targetted molecule present in or around the cell nucleus, the protein or gene symbol that identifies the antibody target that is used as the nuclear marker. This symbol must match the antibody target that is either generated from the panel used or entered with custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets this is the stain name (e.g., DAPI) and, when appropriate, associated staining kit and vendor. For the PhenoCycler, this symbol must match the value found in the XPD output file. | ```DAPI``` ```Not applicable``` | True | -| cell_boundary_marker_or_stain | Allowable Value | If a marker or stain was used to identify all cell boundaries in the tissue, then the name of the marker or stain should be included here. The name of the antibody-targeted molecule marker or non-antibody targeted molecule stain included here must be identical to what is found in the imaging data. For example, with the PhenoCycler, this name must match the value found in the XPD output file. If multiple marker or stains are used to identify all cell boundaries, then a comma separated list should be used here. | ```NAKATPASE``` ```CD298``` ```Not applicable``` | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | diff --git a/docs/assays/metadata/Pixel-seqV2.md b/docs/assays/metadata/Pixel-seqV2.md deleted file mode 100644 index 44e5a345..00000000 --- a/docs/assays/metadata/Pixel-seqV2.md +++ /dev/null @@ -1,37 +0,0 @@ ---- -layout: page ---- -# Pixel-seqV2 - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, this is the area of the FOV (aka ROI) region being captured. For Xenium this is the total area of the FOV regions (aka ROI) being captured. For Stereo-Seq this is the number of beads. | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2```, ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Assigned Value | The unit for spot size value. | ```um^2```, ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | True | -| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| number_of_additional_stains | Numeric | This would be minimally 2 (always include DAPI and polyT) and can include 6 more. | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/testing/RNAseq-(with-probes).md b/docs/assays/metadata/RNAseq-(with-probes).md similarity index 100% rename from docs/assays/metadata/testing/RNAseq-(with-probes).md rename to docs/assays/metadata/RNAseq-(with-probes).md diff --git a/docs/assays/metadata/RNAseq.md b/docs/assays/metadata/RNAseq.md index 8a07abc0..0592557c 100644 --- a/docs/assays/metadata/RNAseq.md +++ b/docs/assays/metadata/RNAseq.md @@ -1,425 +1,95 @@ ---- -layout: page ---- -# RNAseq - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
RNAseq Version 5 (current) - -## RNAseq Version 5 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | True | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | True | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | -| number_of_iterations_of_cdna_amplification | Numeric | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,78``` ```10,48,86``` ```Not applicable``` | True | -| amount_of_input_analyte_unit | Allowable Value | Units of amount of entity input to assay value | ```ug``` ```ng``` | False | - -
- -
RNAseq Version 2 - -## RNAseq Version 2 - -| Attribute | Type | Description | Allowable Values | required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | True | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | True | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | -| amount_of_input_analyte_unit | Textfield | Units of amount of entity input to assay value | | False | -| number_of_iterations_of_cdna_amplification | Numeric | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,78``` ```10,48,86``` ```Not applicable``` | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq (bulk)``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```PhenoCycler``` ```RNAseq (bulk)``` ```scATACseq``` ```scRNAseq``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```snATACseq``` ```snRNAseq``` ```Thick section Multiphoton MxIF``` ```Visium``` ```Xenium``` | True | - -
- -
bulk-RNA Version 1 - -## bulk-RNA Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['bulkATACseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| bulk_transposition_input_number_nuclei | Textfield | A number (no comma separators) | | True | -| bulk_atac_cell_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | True | -| is_technical_replicate | Allowable Value | Is this a sequencing replicate? | ['Yes','No']] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library_concentration_value | ['nM'] | False | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_creation_date | Datetime | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | False | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_pcr_cycles | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | True | -| library_preparation_kit | Textfield | Reagent kit used for library preparation | | True | -| sample_quality_metric | Textfield | This is a quality metric by visual inspection. This should answerthe question: Are the nuclei intact and are the nuclei free of significant amountsof debris? This can be captured at a high level, “OK” or “notOK”. | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | -| transposition_method | Textfield | Modality of capturing accessible chromatin molecules. The kit used, for example. | | True | -| transposition_transposase_source | Textfield | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
bulk-RNA Version 0 - -## bulk-RNA Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['bulk-RNA'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| bulk_rna_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How was tissue stored and processed for RNA isolation RNA_isolation_protocols_io_doi | | True | -| bulk_rna_yield_value | Numeric | RNA (ng) per Weight of Tissue (mg). Answer the question: How much RNA in ng was isolated? How much tissue in mg was initially used for isolating RNA? Calculate the yield by dividing total RNA isolated by amount of tissue used to isolate RNA from (ng/mg). | | True | -| bulk_rna_yield_units_per_tissue_unit | Allowable Value | RNA amount per Tissue input amount. Valid values should be weight/weight (ng/mg). | ['ng/mg'] | True | -| bulk_rna_isolation_quality_metric_value | Numeric | RIN value | | True | -| rnaseq_assay_input_value | Numeric | RNA input amount value to the assay | | True | -| rnaseq_assay_input_unit | Allowable Value | Units of RNA input amount to the assay | ['ug'] | False | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming. | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
scRNAseq Version 3 - -## scRNAseq Version 3 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['3'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The UMI sequence length in the 10xGenomics-v2 kit is 10 base pairs and the length in the 10xGenomics-v3 kit is 12 base pairs. | ['scRNAseq-10xGenomics-v2', 'scRNAseq-10xGenomics-v3', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, TRUE or FALSE | ['Yes','No']] | True | -| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). | | False | -| umi_read | Textfield | Which read file(s) contains the UMI (unique molecular identifier) barcode. | | True | -| umi_offset | Numeric | Position in the read at which the umi barcode starts. | | True | -| umi_size | Numeric | Length of the umi barcode in base pairs. | | True | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | False | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | False | -| expected_cell_count | Numeric | How many cells are expected? This may be used in downstream pipelines to guide selection of cell barcodes or segmentation parameters. | | False | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
scRNAseq Version 2 - -## scRNAseq Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics-v2', 'scRNAseq-10xGenomics-v3', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, TRUE or FALSE | ['Yes','No']] | True | -| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). | | False | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | False | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | False | -| expected_cell_count | Numeric | How many cells are expected? This may be used in downstream pipelines to guide selection of cell barcodes or segmentation parameters. | | False | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
scRNAseq Version 1 - -## scRNAseq Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in repliucate, TRUE or FALSE | ['Yes','No']] | True | -| cell_barcode_read | Textfield | Which read file contains the cell barcode | | True | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | True | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | True | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
scRNAseq Version 0 - -## scRNAseq Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics-v2', 'scRNAseq-10xGenomics-v3', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, TRUE or FALSE | ['Yes','No']] | True | -| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). | | False | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | False | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | False | -| expected_cell_count | Numeric | How many cells are expected? This may be used in downstream pipelines to guide selection of cell barcodes or segmentation parameters. | | False | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
\ No newline at end of file +--- +layout: page-triary +--- + +# RNAseq Metadata Attributes + +Fields that are collected for RNAseq data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id | | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi | | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | ```https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1``` | +| dataset_type | | The specific type of dataset being produced. | | +| analyte_class | | Analytes are the target molecules being measured with the assay. | | +| is_targeted | | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor | | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | +| acquisition_instrument_model | | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | +| source_storage_duration_value | | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit | | The time duration unit of measurement | | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | | +| contributors_path | | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path | | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| barcode_offset | | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | | +| barcode_read | | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | | +| barcode_size | | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | | +| umi_offset | | Position in the read at which the umi barcode starts. | | +| umi_read | | Which read file(s) contains the UMI (unique molecular identifier) barcode. | | +| umi_size | | Length of the umi barcode in base pairs. | | +| assay_input_entity | | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | | +| number_of_input_cells_or_nuclei | | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | +| amount_of_input_analyte_value | | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | +| amount_of_input_analyte_unit | | Units of amount of entity input to assay value | | +| library_adapter_sequence | | Adapter sequence to be used for adapter trimming | | +| library_average_fragment_size | | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | +| library_input_amount_value | | The amount of cDNA, after amplification, that was used for library construction. | | +| library_input_amount_unit | | unit of library input amount value | | +| library_output_amount_value | | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | +| library_output_amount_unit | | Units of library final yield. | | +| library_concentration_value | | The concentration value of the pooled library samples submitted for sequencing. | | +| library_concentration_unit | | Unit of library concentration value. | | +| library_layout | | State whether the library was generated for single-end or paired end sequencing. | | +| number_of_iterations_of_cdna_amplification | | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | +| number_of_pcr_cycles_for_indexing | | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | +| library_preparation_kit | | Reagent kit used for library preparation | | +| sample_indexing_kit | | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | | +| sample_indexing_set | | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | +| is_technical_replicate | | Is the sequencing reaction run in replicate, TRUE or FALSE | | +| expected_entity_capture_count | | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | +| sequencing_reagent_kit | | Reagent kit used for sequencing | | +| sequencing_read_format | | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | +| sequencing_batch_id | | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| capture_batch_id | | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | +| preparation_instrument_kit | | The reagent kit used with the preparation instrument. | | +| metadata_schema_id | | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes +  + + indicates a field that was previously required + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | | +| bulk_rna_isolation_protocols_io_doi | | Link to a protocols document answering the question: How was tissue stored and processed for RNA isolation RNA_isolation_protocols_io_doi | | +| bulk_rna_isolation_quality_metric_value | | RIN value | | +| bulk_rna_yield_units_per_tissue_unit | | RNA amount per Tissue input amount. Valid values should be weight/weight (ng/mg). | | +| bulk_rna_yield_value | | RNA (ng) per Weight of Tissue (mg). Answer the question: How much RNA in ng was isolated? How much tissue in mg was initially used for isolating RNA? Calculate the yield by dividing total RNA isolated by amount of tissue used to isolate RNA from (ng/mg). | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| library_construction_protocols_io_doi | | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | +| library_id | | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| rnaseq_assay_method | | The kit used for the RNA sequencing assay | | +| sc_isolation_enrichment | | The method by which specific cell populations are sorted or enriched. | | +| sc_isolation_protocols_io_doi | | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | +| sc_isolation_quality_metric | | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | +| sc_isolation_tissue_dissociation | | The method by which tissues are dissociated into single cells in suspension. | | +| sc_isolation_cell_number | | Total number of cell/nuclei yielded post dissociation and enrichment | | +| sequencing_phix_percent | | Percent PhiX loaded to the run | | +| sequencing_read_percent_q30 | | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | +| version | | Version of the schema to use when validating this metadata. | | +| description | | Free-text description of this assay. | | diff --git a/docs/assays/metadata/RNAseqWithProbes.md b/docs/assays/metadata/RNAseqWithProbes.md deleted file mode 100644 index fb8e9742..00000000 --- a/docs/assays/metadata/RNAseqWithProbes.md +++ /dev/null @@ -1,63 +0,0 @@ ---- -layout: page ---- -# RNAseq-(with-probes) -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```8,8``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```14``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | False | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | False | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | False | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| False | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | -| number_of_iterations_of_cdna_amplification | Numeric | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```34``` ```36``` ```Not applicable``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | True | -| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ```Yes``` ```No``` | True | -| probe_hybridization_time_value | Numeric | How long was the oligo-conjugated RNA or oligo-conjugated antibody probes hybridized with the sample? | | True | -| probe_hybridization_time_unit | Allowable Value | The units for probe hybridization time value. | ```Hour``` ```Minute``` | True | -| oligo_probe_panel | Allowable Value | This is the probe panel used to target genes and/or proteins. In cases where there is a core panel and add-on modules, the core panel should be selected here. If additional panels are used, then they must be included in the "additional_panels_used.csv" file that's uploaded with the dataset. | ```10x Genomics; Chromium Fixed RNA Kit``` ```Human Transcriptome``` ```4 rxns x 1 BC; PN 1000474``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```16 rxns; PN 1000420``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```64 rxns; PN 1000456``` ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363``` ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365``` ```Custom``` ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-HuWTA-4``` ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-MsWTA-4``` | True | -| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ```Yes``` ```No``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| amount_of_input_analyte_unit | Allowable Value | Units of amount of entity input to assay value | ```ug``` ```ng``` | False | - -
diff --git a/docs/assays/metadata/Raman-Imaging.md b/docs/assays/metadata/Raman-Imaging.md deleted file mode 100644 index 95665aef..00000000 --- a/docs/assays/metadata/Raman-Imaging.md +++ /dev/null @@ -1,45 +0,0 @@ ---- -layout: page ---- -# Raman-Imaging - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| is_image_preprocessing_required | Radio | Indicates whether image preprocessing is necessary based on the type of acquisition instrument used, such as a microscope or slide scanner. This may involve steps like fusing image tiles to assemble the complete image. Example: Yes | ```Yes```, ```No``` | False | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | False | -| tiled_image_columns | Numeric | The number of columns used in the stitching process of a tiled image, often referred to as the grid size in the x-dimension. Example: 5 | | False | -| tiled_image_count | Numeric | The total number of raw tiled images captured, which are intended to be stitched together. Example: 75 | | False | -| intended_tile_overlap_percentage | Numeric | The intended percentage of overlap between tiled images. This value serves as the set point, although slight variations may occur during image acquisition due to stage registration. Example: 5 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq```, ```PhenoCycler``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| tile_configuration | Assigned Value | The configuration of tiles used for stitching in the assay process. If no tile configuration is applicable, enter "Not applicable". Example: Row-by-row | ```Column-by-column```, ```Not applicable```, ```Snake-by-columns```, ```Row-by-row```, ```Snake-by-rows``` | False | -| scan_direction | Assigned Value | The direction of imaging, which is necessary for the stitching process. Example: Left-and-down | ```Left-and-down```, ```Right-and-down```, ```Not applicable```, ```Right-and-up```, ```Left-and-up``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| number_of_pixels | Numeric | The total number of spatial sampling points in an image; for example, in a Raman image, each pixel corresponds to one recorded Raman spectrum. Example: 40000 | | True | -| pixel_physical_size_height_value | Numeric | The physical height of a single pixel in the image. Example: 1000 | | True | -| pixel_physical_size_height_unit | Assigned Value | The unit of measurement for the pixel physical size height value. If the pixel height is not specified, this field may be left blank. Example: um | ```um```, ```mm```, ```nm``` | True | -| pixel_physical_size_width_value | Numeric | The physical width of a single pixel in the image. Example: 1000 | | True | -| pixel_physical_size_width_unit | Assigned Value | The unit of measurement for the pixel physical size width value. If the pixel width value is not specified, this field may be left blank. Example: um | ```um```, ```mm```, ```nm``` | True | -| pixel_physical_size_depth_value | Numeric | The physical depth of a single pixel in the image. Example: 10 | | True | -| pixel_physical_size_depth_unit | Assigned Value | The unit of measurement for the pixel physical size depth value. If the pixel depth value is not specified, this field may be left blank. Example: um | ```um```, ```mm```, ```nm``` | True | -| objective_numerical_aperture | Numeric | Numerical aperture of the microscope objective used to focus the excitation laser on the sample and collect the resulting scattered signal, such as Raman-scattered light. Example: 0.5 | | True | -| laser_power | Numeric | Power of the excitation laser at the sample’s focal plane, measured after the objective and reported in milliwatts (mW). Example: 10 | | True | -| raman_shift_range | Textfield | Range of Raman shifts acquired in the measurement, expressed in wavenumbers (cm⁻¹). Example: 400-3200 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/SIMS.md b/docs/assays/metadata/SIMS.md deleted file mode 100644 index 3880a74d..00000000 --- a/docs/assays/metadata/SIMS.md +++ /dev/null @@ -1,39 +0,0 @@ ---- -layout: page ---- -# SIMS - -
Version 2 (latest) - -## Version 2 (latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | True | -| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | -| matrix_deposition_method | Allowable Value | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| preparation_matrix | Allowable Value | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | -| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
diff --git a/docs/assays/metadata/STARmap.md b/docs/assays/metadata/STARmap.md deleted file mode 100644 index 030cdade..00000000 --- a/docs/assays/metadata/STARmap.md +++ /dev/null @@ -1,43 +0,0 @@ ---- -layout: page ---- -# STARmap - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq```, ```PhenoCycler``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | True | -| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | -| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| proteinasek_concentration | Numeric | The concentration of the enzyme Proteinase K within a sample, measured in micrograms per milliliter (ug/ml). Example: 10 | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is incubated with Proteinase K. Example: 15 | | False | -| proteinasek_incubation_time_unit | Assigned Value | The unit of measurement for the proteinaseK incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | -| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | -| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | -| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | -| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | -| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/SecondHarmonicGeneration.md b/docs/assays/metadata/SecondHarmonicGeneration.md deleted file mode 100644 index 22b4169b..00000000 --- a/docs/assays/metadata/SecondHarmonicGeneration.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# SGH -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
diff --git a/docs/assays/metadata/Seq-Scope.md b/docs/assays/metadata/Seq-Scope.md deleted file mode 100644 index e8e33049..00000000 --- a/docs/assays/metadata/Seq-Scope.md +++ /dev/null @@ -1,39 +0,0 @@ ---- -layout: page ---- -# Seq-Scope - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, this is the area of the FOV (aka ROI) region being captured. For Xenium this is the total area of the FOV regions (aka ROI) being captured. For Stereo-Seq this is the number of beads. | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2```, ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Assigned Value | The unit for spot size value. | ```um^2```, ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Assigned Value | Units corresponding to inter-spot distance | ```um``` | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| number_of_additional_stains | Numeric | This would be minimally 2 (always include DAPI and polyT) and can include 6 more. | | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/Slide-seq.md b/docs/assays/metadata/Slide-seq.md deleted file mode 100644 index 3f7fad31..00000000 --- a/docs/assays/metadata/Slide-seq.md +++ /dev/null @@ -1,94 +0,0 @@ ---- -layout: page ---- -# SnareSeq2 - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 1 (no longer accepting data) - -## Version 1 (no longer accepting data) - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Slide-seq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes', 'No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| puck_id | Textfield | Slide-seq captures RNA sequence data on spatially barcoded arrays of beads. Beads are fixed to a slide in a region shaped like a round puck. Each puck has a unique puck_id. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in repliucate, TRUE or FALSE | ['Yes', 'No'] | True | -| bead_barcode_read | Textfield | Which read file contains the bead barcode | | True | -| bead_barcode_offset | Textfield | Position(s) in the read at which the bead barcode starts | | True | -| bead_barcode_size | Textfield | Length of the bead barcode in base pairs | | True | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Slide-seq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes', 'No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| puck_id | Textfield | Slide-seq captures RNA sequence data on spatially barcoded arrays of beads. Beads are fixed to a slide in a region shaped like a round puck. Each puck has a unique puck_id. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in repliucate, TRUE or FALSE | ['Yes', 'No'] | True | -| bead_barcode_read | Textfield | Which read file contains the bead barcode | | True | -| bead_barcode_offset | Textfield | Position(s) in the read at which the bead barcode starts | | True | -| bead_barcode_size | Textfield | Length of the bead barcode in base pairs | | True | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/SnareSeq2.md b/docs/assays/metadata/SnareSeq2.md deleted file mode 100644 index 2ce1a047..00000000 --- a/docs/assays/metadata/SnareSeq2.md +++ /dev/null @@ -1,19 +0,0 @@ ---- -layout: page ---- -# SnareSeq2 -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|----------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| number_of_pre-amplification_pcr_cycles | Numeric | The number of PCR cycles run after the Chromium Controller step and prior to separating the suspension and initiating library construction | | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/ThickSectionMultiphotonMxIF.md b/docs/assays/metadata/ThickSectionMultiphotonMxIF.md deleted file mode 100644 index a6238240..00000000 --- a/docs/assays/metadata/ThickSectionMultiphotonMxIF.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# MxIF -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
diff --git a/docs/assays/metadata/testing/Visium-(no-probes).md b/docs/assays/metadata/Visium-(no-probes).md similarity index 100% rename from docs/assays/metadata/testing/Visium-(no-probes).md rename to docs/assays/metadata/Visium-(no-probes).md diff --git a/docs/assays/metadata/Visium-HD.md b/docs/assays/metadata/Visium-HD.md deleted file mode 100644 index c1cae516..00000000 --- a/docs/assays/metadata/Visium-HD.md +++ /dev/null @@ -1,33 +0,0 @@ ---- -layout: page ---- -# Visium-HD - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, this is the area of the FOV (aka ROI) region being captured. For Xenium this is the total area of the FOV regions (aka ROI) being captured. For Stereo-Seq this is the number of beads. | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2```, ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Assigned Value | The unit for spot size value. | ```um^2```, ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Assigned Value | Units corresponding to inter-spot distance | ```um``` | True | -| capture_area_id | Radio | Which capture area on the slide was used. For Visium this would be [A1, B1, C1, D1]. For HiFi this would be the lane on the flowcell. | ```A1```, ```B1```, ```C1```, ```D1```, ```Lane 1```, ```Lane 2```, ```Lane 3```, ```Lane 4```, ```Lane 5```, ```Lane 6```, ```Lane 7```, ```Lane 8``` | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| preparation_instrument_vendor | Assigned Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```Thermo Fisher Scientific```, ```SunChrom```, ```Leica Biosystems```, ```Roche Diagnostics```, ```In-House```, ```Not applicable```, ```Hamamatsu```, ```HTX Technologies```, ```10x Genomics``` | True | -| preparation_instrument_model | Assigned Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL```, ```ST5020 Multistainer```, ```Visium CytAssist```, ```SunCollect Sprayer```, ```Chromium X```, ```Chromium iX```, ```EVOS M7000```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```Discovery Ultra```, ```Sublimator```, ```Not applicable```, ```TM-Sprayer```, ```M5 Sprayer```, ```M3+ Sprayer```, ```Chromium Controller```, ```Chromium Connect``` | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/VisiumNoProbes.md b/docs/assays/metadata/VisiumNoProbes.md deleted file mode 100644 index 2d741c42..00000000 --- a/docs/assays/metadata/VisiumNoProbes.md +++ /dev/null @@ -1,58 +0,0 @@ ---- -layout: page ---- -# Visium-(no-probes) - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version3 (current) - -## Version 3 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Allowable Value | The unit for spot size value. | ```um^2``` ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Allowable Value | Units corresponding to inter-spot distance | ```um``` | True | -| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be ```A1, B1, C1, D1```. For HiFi this would be the lane on the flowcell. | ```A1``` ```B1``` ```C1``` ```D1``` ```Lane 1``` ```Lane 2``` ```Lane 3``` ```Lane 4``` ```Lane 5``` ```Lane 6``` ```Lane 7``` ```Lane 8``` | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | - -
- -
Version 2 - -## Version 2 - -| Attribute | Type | Description | Allowable Value | Required | -|-----------------------------|----------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------|------------| -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| dataset_type | Textfield | The specific type of dataset being produced. | | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Textfield | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Textfield | The unit for spot size value. | | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Textfield | Units corresponding to inter-spot distance | | True | -| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be [A1, B1, C1, D1]. For HiFi this would be the lane on the flowcell. | [A1, B1, C1, D1, Lane 1, Lane 2, Lane 3, Lane 4, Lane 5, Lane 6, Lane 7, Lane 8] | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Textfield | The unit for the permeabilization time. | | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/VisiumWithProbes.md b/docs/assays/metadata/VisiumWithProbes.md deleted file mode 100644 index 89c07f31..00000000 --- a/docs/assays/metadata/VisiumWithProbes.md +++ /dev/null @@ -1,30 +0,0 @@ ---- -layout: page ---- -# Visium-(with-probes) -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Allowable Value | The unit for spot size value. | ```um^2``` ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Allowable Value | Units corresponding to inter-spot distance | ```um``` | True | -| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be ```A1, B1, C1, D1```. For HiFi this would be the lane on the flowcell. | ```A1``` ```B1``` ```C1``` ```D1``` ```Lane 1``` ```Lane 2``` ```Lane 3``` ```Lane 4``` ```Lane 5``` ```Lane 6``` ```Lane 7``` ```Lane 8``` | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | True | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| True | - -
diff --git a/docs/assays/metadata/WGS.md b/docs/assays/metadata/WGS.md deleted file mode 100644 index 20a9ee8b..00000000 --- a/docs/assays/metadata/WGS.md +++ /dev/null @@ -1,86 +0,0 @@ ---- -layout: page ---- -# WGS - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 1 (no longer accepting data) - -## Version 1 (no longer accepting data) - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['WGS'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| gdna_fragmentation_quality_assurance | Allowable Value | Is the gDNA integrity good enough for WGS? This is usually checked through running a gel. | ['Pass', 'Fail'] | True | -| dna_assay_input_value | Numeric | Amount of DNA input into library preparation | | True | -| dna_assay_input_unit | Allowable Value | Units of DNA input into library preparation | ['ug'] | False | -| library_construction_method | Textfield | Describes DNA library preparation kit. Modality of isolating gDNA, Fragmentation and generating sequencing libraries. | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | The adapter sequence to be used for adapter trimming starting with the 5' end. (eg. 5-ATCCTGAGAA) | | True | -| library_final_yield | Numeric | Total amount of library after final pcr amplification step | | True | -| library_final_yield_unit | Allowable Value | Total units of library after final pcr amplification step | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['WGS'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| gdna_fragmentation_quality_assurance | Allowable Value | Is the gDNA integrity good enough for WGS? This is usually checked through running a gel. | ['Pass', 'Fail'] | True | -| dna_assay_input_value | Numeric | Amount of DNA input into library preparation | | True | -| dna_assay_input_unit | Allowable Value | Units of DNA input into library preparation | ['ug'] | False | -| library_construction_method | Textfield | Describes DNA library preparation kit. Modality of isolating gDNA, Fragmentation and generating sequencing libraries. | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | The adapter sequence to be used for adapter trimming starting with the 5' end. (eg. 5-ATCCTGAGAA) | | True | -| library_final_yield | Numeric | Total amount of library after final pcr amplification step | | True | -| library_final_yield_unit | Allowable Value | Total units of library after final pcr amplification step | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/testing/comet.md b/docs/assays/metadata/comet.md similarity index 100% rename from docs/assays/metadata/testing/comet.md rename to docs/assays/metadata/comet.md diff --git a/docs/assays/metadata/testing/cosmx-proteomics.md b/docs/assays/metadata/cosmx-proteomics.md similarity index 100% rename from docs/assays/metadata/testing/cosmx-proteomics.md rename to docs/assays/metadata/cosmx-proteomics.md diff --git a/docs/assays/metadata/testing/cosmx-transcriptomics.md b/docs/assays/metadata/cosmx-transcriptomics.md similarity index 100% rename from docs/assays/metadata/testing/cosmx-transcriptomics.md rename to docs/assays/metadata/cosmx-transcriptomics.md diff --git a/docs/assays/metadata/testing/cycif.md b/docs/assays/metadata/cycif.md similarity index 100% rename from docs/assays/metadata/testing/cycif.md rename to docs/assays/metadata/cycif.md diff --git a/docs/assays/metadata/testing/cytof.md b/docs/assays/metadata/cytof.md similarity index 100% rename from docs/assays/metadata/testing/cytof.md rename to docs/assays/metadata/cytof.md diff --git a/docs/assays/metadata/testing/dna-methylation.md b/docs/assays/metadata/dna-methylation.md similarity index 100% rename from docs/assays/metadata/testing/dna-methylation.md rename to docs/assays/metadata/dna-methylation.md diff --git a/docs/assays/metadata/testing/enhancedsrs.md b/docs/assays/metadata/enhancedsrs.md similarity index 100% rename from docs/assays/metadata/testing/enhancedsrs.md rename to docs/assays/metadata/enhancedsrs.md diff --git a/docs/assays/metadata/testing/facs.md b/docs/assays/metadata/facs.md similarity index 100% rename from docs/assays/metadata/testing/facs.md rename to docs/assays/metadata/facs.md diff --git a/docs/assays/metadata/testing/geomx.md b/docs/assays/metadata/geomx.md similarity index 100% rename from docs/assays/metadata/testing/geomx.md rename to docs/assays/metadata/geomx.md diff --git a/docs/assays/metadata/testing/hifi.md b/docs/assays/metadata/hifi.md similarity index 100% rename from docs/assays/metadata/testing/hifi.md rename to docs/assays/metadata/hifi.md diff --git a/docs/assays/metadata/iCLAP.md b/docs/assays/metadata/iCLAP.md deleted file mode 100644 index ce13e526..00000000 --- a/docs/assays/metadata/iCLAP.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# iCLAP - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| number_of_antibodies | Numeric | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | | True | -| number_of_biomarker_imaging_rounds | Numeric | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | | True | -| number_of_total_imaging_rounds | Numeric | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | | True | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| antibodies_path | Textfield | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| number_of_channels | Numeric | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/testing/iclap.md b/docs/assays/metadata/iclap.md similarity index 100% rename from docs/assays/metadata/testing/iclap.md rename to docs/assays/metadata/iclap.md diff --git a/docs/assays/metadata/testing/illumina-spatial.md b/docs/assays/metadata/illumina-spatial.md similarity index 100% rename from docs/assays/metadata/testing/illumina-spatial.md rename to docs/assays/metadata/illumina-spatial.md diff --git a/docs/assays/metadata/testing/imc.md b/docs/assays/metadata/imc.md similarity index 100% rename from docs/assays/metadata/testing/imc.md rename to docs/assays/metadata/imc.md diff --git a/docs/assays/metadata/index.md b/docs/assays/metadata/index.md index ddda0ecb..efa18772 100644 --- a/docs/assays/metadata/index.md +++ b/docs/assays/metadata/index.md @@ -3,8 +3,7 @@ layout: page --- ## HuBMAP Metadata by Dataset Type -A list of available dataset types (data types from multiple supported assays), with a link [](EnhancedSRS "Attribute description") to the valid metadata attributes for each dataset type. The linked assay metadata pages list all attributes, as they have occurred, across any versions of the metadata specification for the given dataset type with the most current, valid set of attributes listed first on the page. The directory schema for each dataset type is also linked in the description column. - +A list of available dataset types (data types from multiple supported assays), with a link [](EnhancedSRS "Attribute description") to the valid metadata attributes for each dataset type. The linked assay metadata pages list all attributes, as they have occurred, across any versions of the metadata specification for the given dataset type with the most current, valid set of attributes listed first on the page. The directory schema for each dataset type is also linked in the description column. | Dataset Type | Description | |--------------|-------------| @@ -26,24 +25,24 @@ A list of available dataset types (data types from multiple supported assays), w | [IMC](https://docs.hubmapconsortium.org/assays/imc) [](IMC "Attribute description")| Combines standard immunohistochemistry with CyTOF mass cytometry to resolve the cellular localization of up to 40 proteins in a tissue sample. Link to [IMC directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/imc-2d/current/). | | [LC-MS](https://docs.hubmapconsortium.org/assays/lcms) [](LC-MS "Attribute description")| Coupling of liquid chromatography (LC) to mass spectrometry (MS). Link to [LC-MS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/lcms/current/). | | [Light Sheet](https://en.wikipedia.org/wiki/Light_sheet_fluorescence_microscopy) [](LightSheet "Attribute description")| A fluorescence imaging technique that uses a thin sheet of laser light to illuminate a sample, allowing for high-resolution, 3D imaging with reduced photobleaching and phototoxicity; particularly useful for imaging large, thick, or delicate biological samples, like developing embryos or organoids. Link to [Light Sheet directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/lightsheet/current/). | -| [MALDI-IMS](https://docs.hubmapconsortium.org/assays/maldi-ims) [](MALDI "Attribute description") | Matrix-assisted laser desorption/ionization (MALDI) imaging mass spectrometry (IMS) combines the sensitivity and molecular specificity of MS with the spatial fidelity of classical microscopy. Link to [MALDI-IMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/maldi/current/). | -| [MIBI](https://www.researchgate.net/figure/Multiplexed-ion-beam-imaging-workflow-for-high-resolution-spatial-proteomics-Here_fig1_349770840) [](MIBI "Attribute description") | Preserved tissue sections, mounted on conductive substrates are incubated with unique isotopic transition metal-tagged antibody reporters. An oxygen primary ion beam rasters the sample surface, ejecting and ionizing the isotope reporters. Their masses are subsequently measured via a mass analyzer. Link to [MIBI directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/mibi/current/). | -| [MERFISH](https://pubmed.ncbi.nlm.nih.gov/27241748/) [](MERFISH "Attribute description") | A spatial transcriptomics technology that allows for the simultaneous imaging of hundreds to thousands of RNA species within single cells, providing both copy number and spatial distribution information. Link to [MERFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/merfish/current/). | -| [MUSIC](https://www.nature.com/articles/s41586-024-07239-w) [](MUSIC "Attribute description") | A sequencing assay that allows profiling of gene expression, co-complexed DNA sequences, and RNA-chromatin interactions from the same single-cell nucleus. Both RNA and fragmented DNA within a nucleus are labelled with a unique cell barcode, enabling identification and matching of RNA and DNA sequences. Link to [MUSIC directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/music/current/). | -| [MxIF](https://pmc.ncbi.nlm.nih.gov/articles/PMC9959383/#) [](ThickSectionMultiphotonMxIF "Attribute description") | One version of MXIF (multiplexed fluorescence microscopy), an imaging platform whereby a large number of cellular and histological markers can be investigated on a single tissue section. Link to [TSM MxIF directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/thick-section-multiphoton-mxif/current/).| -| Pixel-seqV2 [](Pixel-seqV2 "Attribute description") | Pixel-seqV2 is a spatial transcriptomics method that utilizes polony gels to capture and sequence RNA, proteins or other molecules in tissues with high resolution. These polony gels are arrays of micron-scale DNA clusters, each containing a unique barcode, allowing for the mapping of molecules within their original spatial context in a tissue, thereby allowing researchers to study the spatial organization of cells and their gene expression profiles within tissues.| -| [RNAseq](https://docs.hubmapconsortium.org/assays/rnaseq) [](RNAseq "Attribute description") | While bulk RNAseq elucidates the average gene expression profile in cells comprising a tissue sample, single-cell RNAseq employs per-cell and per-molecule barcoding to enable single-cell resolution of the gene expression profile. Link to [RNAseq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq/current/).| -| [RNAseq with Probes](https://pmc.ncbi.nlm.nih.gov/articles/PMC5717752/#) [](RNAseqWithProbes "Attribute description") | Uses probes to capture and enrich specific regions of the RNA for targeted sequencing, allowing for in-depth analysis of those regions. Link to [RNAseq with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq-with-probes/current/).| -| Raman-Imaging [](Raman-Imaging "Attribute description") | Raman Imaging is a non-invasive technique that maps the unique chemical fingerprint of biological samples (cells, tissues) by capturing Raman scattering (light interacting with molecular vibrations) at each pixel, creating detailed molecular maps showing the distribution of proteins, lipids, DNA, and water. Link to [Raman Imaging directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/raman-imaging/current/). | -| [SHG](https://en.wikipedia.org/wiki/Second-harmonic_imaging_microscopy) [](SecondHarmonicGeneration "Attribute description") | Single-cycle Fluorescence Microscopy (SFM). A technique that utilizes the nonlinear optical phenomenon of SHG to image biological tissues and structures, particularly those containing collagen. Link to [SHG directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/second-harmonic-generation/current/).| -| [SeqFISH](https://docs.hubmapconsortium.org/assays/seqfish) [](seqFISH "Attribute description") | SeqFISH technology allows in situ imaging of multiple mRNAs using barcoding and fluorophore-labelled barcode readout-probes. Link to [SeqFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/seqfish/). _The consortium is no longer accepting data of this type_.| -| [SIMS](https://www.frontiersin.org/journals/chemistry/articles/10.3389/fchem.2023.1237408/full) [](SIMS "Attribute description") | Secondary-ion mass spectrometry (SIMS) is a technique used to analyze the composition of solid surfaces and thin films by sputtering the surface of the specimen. Link to [SIMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/sims/current/ "directory schema").| -| [Slide-seq](https://www.nature.com/articles/s41587-020-0739-1) [](Slide-seq "Attribute description") | Provides a scalable method for obtaining spatially resolved gene expression data at resolutions comparable to the sizes of individual cells. Link to [Slide-seq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/slide-seq/current/).| -| [SnareSeq2](https://www.nature.com/articles/s41596-021-00507-3) [](SnareSeq2 "Attribute description") | This method uses tagmentation within permeabilized and fixed single-nucleus isolates to capture accessible chromatin (AC) regions, followed by the capture and reverse transcription of RNA transcripts. Link to [SnareSeq2 directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/snareseq2/current/).| -| STARmap [](STARmap "Attribute description") | STARmap (Spatially-resolved Transcript Amplicon Readout Mapping) is a biomedical technology that enables the 3D mapping of gene expression within intact tissues at single-cell resolution. It combines hydrogel-tissue chemistry and in situ DNA sequencing to preserve a cell's location and identify which genes are active in that specific spatial context. [STARmap directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/starmap/current/). | -| [Visium No Probes](https://ostr.ccr.cancer.gov/emerging-technologies/spatial-biology/visium/) [](VisiumNoProbes "Attribute description") | A spatial transcriptomics solution that allows researchers to analyze gene expression patterns within the spatial context of a tissue. An in situ method that captures RNA transcripts within the tissue and then sequences them. Link to [Visium NP directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-no-probes/current/). | -| [Visium with Probes](https://ngisweden.scilifelab.se/methods/10x-genomics-visium-cytassist-for-ffpe-samples/) [](VisiumWithProbes "Attribute description") | Offers spatially resolved transcriptomics through the 10X Genomics Visium CytAssist, which combines histology with probe-based transcriptomics in a spatial context. Link to [Visium with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-with-probes/current/). | -| [WGS](https://en.wikipedia.org/wiki/Whole_genome_sequencing) [](WGS "Attribute description") | The process of determining the entire DNA sequence of an organism's genome at a single time. This entails sequencing all of an organism's chromosomal and mitochondrial DNA. Link to [WGS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/wgs/). _The consortium is no longer accepting data of this type_. | +| [MALDI-IMS](https://docs.hubmapconsortium.org/assays/maldi-ims) [](MALDI "Attribute description") | Matrix-assisted laser desorption/ionization (MALDI) imaging mass spectrometry (IMS) combines the sensitivity and molecular specificity of MS with the spatial fidelity of classical microscopy. Link to [MALDI-IMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/maldi/current/). | +| [MIBI](https://www.researchgate.net/figure/Multiplexed-ion-beam-imaging-workflow-for-high-resolution-spatial-proteomics-Here_fig1_349770840) [](MIBI "Attribute description") | Preserved tissue sections, mounted on conductive substrates are incubated with unique isotopic transition metal-tagged antibody reporters. An oxygen primary ion beam rasters the sample surface, ejecting and ionizing the isotope reporters. Their masses are subsequently measured via a mass analyzer. Link to [MIBI directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/mibi/current/). | +| [MERFISH](https://pubmed.ncbi.nlm.nih.gov/27241748/) [](MERFISH "Attribute description") | A spatial transcriptomics technology that allows for the simultaneous imaging of hundreds to thousands of RNA species within single cells, providing both copy number and spatial distribution information. Link to [MERFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/merfish/current/). | +| [MUSIC](https://www.nature.com/articles/s41586-024-07239-w) [](MUSIC "Attribute description") | A sequencing assay that allows profiling of gene expression, co-complexed DNA sequences, and RNA-chromatin interactions from the same single-cell nucleus. Both RNA and fragmented DNA within a nucleus are labelled with a unique cell barcode, enabling identification and matching of RNA and DNA sequences. Link to [MUSIC directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/music/current/). | +| [MxIF](https://pmc.ncbi.nlm.nih.gov/articles/PMC9959383/#) [](ThickSectionMultiphotonMxIF "Attribute description") | One version of MXIF (multiplexed fluorescence microscopy), an imaging platform whereby a large number of cellular and histological markers can be investigated on a single tissue section. Link to [TSM MxIF directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/thick-section-multiphoton-mxif/current/).| +| Pixel-seqV2 [](Pixel-seqV2 "Attribute description") | Pixel-seqV2 is a spatial transcriptomics method that utilizes polony gels to capture and sequence RNA, proteins or other molecules in tissues with high resolution. These polony gels are arrays of micron-scale DNA clusters, each containing a unique barcode, allowing for the mapping of molecules within their original spatial context in a tissue, thereby allowing researchers to study the spatial organization of cells and their gene expression profiles within tissues.| +| [RNAseq](https://docs.hubmapconsortium.org/assays/rnaseq) [](RNAseq "Attribute description") | While bulk RNAseq elucidates the average gene expression profile in cells comprising a tissue sample, single-cell RNAseq employs per-cell and per-molecule barcoding to enable single-cell resolution of the gene expression profile. Link to [RNAseq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq/current/).| +| [RNAseq with Probes](https://pmc.ncbi.nlm.nih.gov/articles/PMC5717752/#) [](RNAseqWithProbes "Attribute description") | Uses probes to capture and enrich specific regions of the RNA for targeted sequencing, allowing for in-depth analysis of those regions. Link to [RNAseq with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq-with-probes/current/).| +| Raman-Imaging [](Raman-Imaging "Attribute description") | Raman Imaging is a non-invasive technique that maps the unique chemical fingerprint of biological samples (cells, tissues) by capturing Raman scattering (light interacting with molecular vibrations) at each pixel, creating detailed molecular maps showing the distribution of proteins, lipids, DNA, and water. Link to [Raman Imaging directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/raman-imaging/current/). | +| [SHG](https://en.wikipedia.org/wiki/Second-harmonic_imaging_microscopy) [](SecondHarmonicGeneration "Attribute description") | Single-cycle Fluorescence Microscopy (SFM). A technique that utilizes the nonlinear optical phenomenon of SHG to image biological tissues and structures, particularly those containing collagen. Link to [SHG directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/second-harmonic-generation/current/).| +| [SeqFISH](https://docs.hubmapconsortium.org/assays/seqfish) [](seqFISH "Attribute description") | SeqFISH technology allows in situ imaging of multiple mRNAs using barcoding and fluorophore-labelled barcode readout-probes. Link to [SeqFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/seqfish/). _The consortium is no longer accepting data of this type_.| +| [SIMS](https://www.frontiersin.org/journals/chemistry/articles/10.3389/fchem.2023.1237408/full) [](SIMS "Attribute description") | Secondary-ion mass spectrometry (SIMS) is a technique used to analyze the composition of solid surfaces and thin films by sputtering the surface of the specimen. Link to [SIMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/sims/current/ "directory schema").| +| [Slide-seq](https://www.nature.com/articles/s41587-020-0739-1) [](Slide-seq "Attribute description") | Provides a scalable method for obtaining spatially resolved gene expression data at resolutions comparable to the sizes of individual cells. Link to [Slide-seq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/slide-seq/current/).| +| [SnareSeq2](https://www.nature.com/articles/s41596-021-00507-3) [](SnareSeq2 "Attribute description") | This method uses tagmentation within permeabilized and fixed single-nucleus isolates to capture accessible chromatin (AC) regions, followed by the capture and reverse transcription of RNA transcripts. Link to [SnareSeq2 directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/snareseq2/current/).| +| STARmap [](STARmap "Attribute description") | STARmap (Spatially-resolved Transcript Amplicon Readout Mapping) is a biomedical technology that enables the 3D mapping of gene expression within intact tissues at single-cell resolution. It combines hydrogel-tissue chemistry and in situ DNA sequencing to preserve a cell's location and identify which genes are active in that specific spatial context. [STARmap directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/starmap/current/). | +| [Visium No Probes](https://ostr.ccr.cancer.gov/emerging-technologies/spatial-biology/visium/) [](VisiumNoProbes "Attribute description") | A spatial transcriptomics solution that allows researchers to analyze gene expression patterns within the spatial context of a tissue. An in situ method that captures RNA transcripts within the tissue and then sequences them. Link to [Visium NP directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-no-probes/current/). | +| [Visium with Probes](https://ngisweden.scilifelab.se/methods/10x-genomics-visium-cytassist-for-ffpe-samples/) [](VisiumWithProbes "Attribute description") | Offers spatially resolved transcriptomics through the 10X Genomics Visium CytAssist, which combines histology with probe-based transcriptomics in a spatial context. Link to [Visium with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-with-probes/current/). | +| [WGS](https://en.wikipedia.org/wiki/Whole_genome_sequencing) [](WGS "Attribute description") | The process of determining the entire DNA sequence of an organism's genome at a single time. This entails sequencing all of an organism's chromosomal and mitochondrial DNA. Link to [WGS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/wgs/). _The consortium is no longer accepting data of this type_. | diff --git a/docs/assays/metadata/testing/info3.png b/docs/assays/metadata/testing/info3.png deleted file mode 100644 index 811c300e2d8c4f154f3319a208424e3e135518ef..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 2038 zcmY*ae>~IqAOFl)Hu;exQgXVuJGafukC|U`JqzzH|GoDQrssKvxJFpL^@7w-dRcJ=Jl3W4c+D|&GG%ZyP<#~i80Dz9Z z+CZn5!4;|tIqZO7elW=!%ZQ6b(3o*_7D5=yQ%wT^ZjX>hV;p7iE$FN$HWzRG>Za7% zg3ZKR2RV>zNxc26Xtqa6Jj*}DCxDT1l;O;@-b2uZ;e=QfLM)3MRxqMV(bk+yb7J3F{)0-lh} z<7~wx1)mv5Bzsjg(`d3{RS0$-HrQNqhB{~40wd6^34T#?VvNR;REYPU? z_~P!;JKo;@Vll$x!_+`+MV*IRns0M?xlt&kaZ5PGw?ixFwQI$xlT%sxu$aJc*HiQk zU9G2nu{P!6Tyb7Bb*wTsw&ml(P2RCy@7twDsik{1qMta;K3oXgmm@k%6?HL$Id$sXa;d!ZP82~YyzFQprp0NEM%_Y>MiA}dx4 z=m`)9eg&kFrHt}z*GaN{3Ya3;^W6#+u)EZ8$hEU8-94478M-zPxv(KD@ifB{IWm~x zk`ZF4yzw#}v3xRgbpAu*FJ3ZDDVp?jXYP%U%Z5UOCF~A+H}mU3V5gg9F5CIXp^4y& zE()s%n@0Zy(?1~I5Ui+F#!+C_=K|}gy#xrZP|y5|;Rr+#R-Bd-I`U+lpb-XYdl(Sb z3@qALq0=7Cy=Xpa{GsmTxI80Abn)S6@m~Gc?)Gxwca>}Jhzt&3PA0E+R3nF;(G~iv zQPGSHjz}|a1_fypvte_T_Zj)JBlmlU&%I?McF1xHI3X11>N9WYz=5zrSiZgw#ZW7ej7ZIk4iEeBDg?)FjliKGBHf}9XeG;1`&;J5tHs2SX#j9Gj0HrOIB z-3cUME5S0N?x*J-R>(@?ZLvXe&VK(6?S9N-VmI=>wPsPfJgoL5M_~F$+<&U^o-*-V&lW9<{!+DifOh_R&^1+Df#7 zhw{D(St-^Fq$p_hd0dW}=mhD({uB;g@*5vLZ*hD3E-|2C$BdT*^wxCY(dre+?%STuFn%6#*8Z0Kp!Z*Br928mWk_wrqX=9fxFXdazr zg$5K|y@aOug;woV&OSdKyX6zBbc;pF_%rWN-Jiq}m1gkhmftW>&dF!oJPErNR&3D5 zTX!9I{5tl`<{D)0U7CiRk$LNs>~7E0{z9ZzN>0p!7f?)D_coDqf+8*-4AjdVYTLMk zFqDsfIW)yAJY%rD@6X3Y^1v+91?ai;n?1J@&wb+D_~Lqr{dS$jw>6iaT6HBU@7B6* zwhbcWFB~ZZNj)V)=W#<{;*hZW8-6w3{)5F55nsx@KiGF;DhNMs3B6 zbsnsn{7ALDo+fK52nS1sMBSaY$mNf%)5GW3U&L|$n353F*GFvMU_#sHAGA)aDvGLU z&dRFcq!o(o;OOLY28tKOv}8;kRs?j%755*V?3X;VE&rObeYZnGSzs$HyveP=&!{!~ z>O|iLZeu?#ScCGaFK(c4Tyxb>a{zYlu;+!ktep?jKmKlDuBrFnF1q*{6OqCwb-Y5W zsd|3?w6bwyo1O8&dvx5My+@S({hPWHhf_L-H5N5b823)MRoA=R8yIBN z@SN>~d&dj{m)gX0%Qd|G5@|#2$zvHGC_lC$1PyG3M%Phkv1v`E&1EeNSVD^Ram@J& zI~=c=kAF9U>r2VaCWOu$Yo{fiNopIK+1ODro!_^~bLJ*3Jm@x~=d}+l@_U!>&tE9* ztniR1OCsJE1xoyd^7tsX>D_r}soIqKBuL@dSqRjDc%XpikEZ<964^TQiIR9KMh+>; zSX?Wk%;5igKW%=swu=pCGQ2&xCN7EJx#g}`=&U?mP3jeLP3Eev{D^&_cJx)9tJkZX T8k8ku_2=v9=0mLC7m@ilud0$r diff --git a/docs/assays/metadata/testing/thicksectionmultiphotonmxif.md b/docs/assays/metadata/thicksectionmultiphotonmxif.md similarity index 100% rename from docs/assays/metadata/testing/thicksectionmultiphotonmxif.md rename to docs/assays/metadata/thicksectionmultiphotonmxif.md diff --git a/docs/assays/metadata/testing/visium-hd.md b/docs/assays/metadata/visium-hd.md similarity index 100% rename from docs/assays/metadata/testing/visium-hd.md rename to docs/assays/metadata/visium-hd.md diff --git a/docs/assays/metadata/testing/visiumwithprobes.md b/docs/assays/metadata/visiumwithprobes.md similarity index 100% rename from docs/assays/metadata/testing/visiumwithprobes.md rename to docs/assays/metadata/visiumwithprobes.md diff --git a/docs/assays/metadata/testing/wgs.md b/docs/assays/metadata/wgs.md similarity index 100% rename from docs/assays/metadata/testing/wgs.md rename to docs/assays/metadata/wgs.md diff --git a/scripts/newMeta2/source/reharmonize-legacy-metadata b/scripts/newMeta2/source/reharmonize-legacy-metadata new file mode 160000 index 00000000..b184c78a --- /dev/null +++ b/scripts/newMeta2/source/reharmonize-legacy-metadata @@ -0,0 +1 @@ +Subproject commit b184c78aa460ff60e961bf3aa4f5de7d0de53406 From 9d97594ed3185ac37dd0b1d5104623d4e8df5343 Mon Sep 17 00:00:00 2001 From: Birdmachine Date: Fri, 17 Jul 2026 09:23:44 -0400 Subject: [PATCH 2/3] Revert "switch assay metadata pages to Harmonized set (& purge holding Testing directory)" This reverts commit 1d5da380fc2f3177e47d5a83e21f0ea0b0bea578. --- docs/assays/metadata/10XMultiome.md | 24 + docs/assays/metadata/4i.md | 48 +- docs/assays/metadata/ATACseq.md | 351 ++++++++---- docs/assays/metadata/AutoFluorescence.md | 105 ++++ docs/assays/metadata/CODEX.md | 182 +++--- docs/assays/metadata/COMET.md | 37 ++ docs/assays/metadata/CosMx-Proteomics.md | 43 ++ docs/assays/metadata/CosMx-Transcriptomics.md | 88 +++ docs/assays/metadata/CyCIF.md | 34 ++ docs/assays/metadata/CyTOF.md | 38 ++ docs/assays/metadata/DESI.md | 112 ++-- docs/assays/metadata/DNA-Methylation.md | 28 + docs/assays/metadata/EnhancedSRS.md | 34 ++ docs/assays/metadata/FACS.md | 39 ++ docs/assays/metadata/GeoMx.md | 97 ++++ docs/assays/metadata/HiFi.md | 47 ++ docs/assays/metadata/Histology.md | 109 ++-- docs/assays/metadata/IMC.md | 234 ++++++++ docs/assays/metadata/Illumina-Spatial.md | 41 ++ docs/assays/metadata/LC-MS.md | 385 ++++++++++--- docs/assays/metadata/LightSheet.md | 146 +++++ docs/assays/metadata/MALDI.md | 238 +++++--- docs/assays/metadata/MERFISH.md | 47 ++ docs/assays/metadata/MIBI.md | 190 ++++--- docs/assays/metadata/MPLEx.md | 59 ++ docs/assays/metadata/MUSIC.md | 128 ++--- docs/assays/metadata/Olink.md | 28 + docs/assays/metadata/PhenoCycler.md | 36 ++ docs/assays/metadata/Pixel-seqV2.md | 37 ++ docs/assays/metadata/RNAseq.md | 520 ++++++++++++++---- docs/assays/metadata/RNAseqWithProbes.md | 63 +++ docs/assays/metadata/Raman-Imaging.md | 45 ++ docs/assays/metadata/SIMS.md | 39 ++ docs/assays/metadata/STARmap.md | 43 ++ .../metadata/SecondHarmonicGeneration.md | 34 ++ docs/assays/metadata/Seq-Scope.md | 39 ++ docs/assays/metadata/Slide-seq.md | 94 ++++ docs/assays/metadata/SnareSeq2.md | 19 + .../metadata/ThickSectionMultiphotonMxIF.md | 34 ++ docs/assays/metadata/Visium-HD.md | 33 ++ docs/assays/metadata/VisiumNoProbes.md | 58 ++ docs/assays/metadata/VisiumWithProbes.md | 30 + docs/assays/metadata/WGS.md | 86 +++ docs/assays/metadata/iCLAP.md | 34 ++ docs/assays/metadata/index.md | 39 +- docs/assays/metadata/link2.png | Bin 0 -> 1449 bytes docs/assays/metadata/seqFISH.md | 90 +++ docs/assays/{ => metadata/testing}/.directory | 2 +- .../metadata/{ => testing}/10X-Multiome.md | 0 docs/assays/metadata/testing/4i.md | 21 + docs/assays/metadata/testing/ATACseq.md | 94 ++++ .../{ => testing}/Auto-fluorescence.md | 0 docs/assays/metadata/testing/CODEX.md | 62 +++ .../metadata/{ => testing}/Cell-DIVE.md | 0 docs/assays/metadata/testing/DESI.md | 69 +++ docs/assays/metadata/testing/Histology.md | 68 +++ docs/assays/metadata/{ => testing}/IMC-2D.md | 0 docs/assays/metadata/testing/LC-MS.md | 89 +++ .../metadata/{ => testing}/Light-Sheet.md | 0 docs/assays/metadata/testing/MALDI.md | 67 +++ docs/assays/metadata/testing/MIBI.md | 86 +++ .../metadata/{ => testing}/MUSIC-(CEDAR).md | 0 docs/assays/metadata/testing/MUSIC.md | 70 +++ .../{ => testing}/RNAseq-(with-probes).md | 0 docs/assays/metadata/testing/RNAseq.md | 95 ++++ .../{ => testing}/Visium-(no-probes).md | 0 docs/assays/metadata/{ => testing}/comet.md | 0 .../{ => testing}/cosmx-proteomics.md | 0 .../{ => testing}/cosmx-transcriptomics.md | 0 docs/assays/metadata/{ => testing}/cycif.md | 0 docs/assays/metadata/{ => testing}/cytof.md | 0 .../metadata/{ => testing}/dna-methylation.md | 0 .../metadata/{ => testing}/enhancedsrs.md | 0 docs/assays/metadata/{ => testing}/facs.md | 0 docs/assays/metadata/{ => testing}/geomx.md | 0 docs/assays/metadata/{ => testing}/hifi.md | 0 docs/assays/metadata/{ => testing}/iclap.md | 0 .../{ => testing}/illumina-spatial.md | 0 docs/assays/metadata/{ => testing}/imc.md | 0 docs/assays/metadata/testing/index.md | 49 ++ docs/assays/metadata/testing/info3.png | Bin 0 -> 2038 bytes docs/assays/metadata/{ => testing}/merfish.md | 0 docs/assays/metadata/{ => testing}/mplex.md | 0 docs/assays/metadata/{ => testing}/olink.md | 0 .../metadata/{ => testing}/phenocycler.md | 0 .../metadata/{ => testing}/pixel-seqv2.md | 0 .../metadata/{ => testing}/raman-imaging.md | 0 .../{ => testing}/secondharmonicgeneration.md | 0 .../metadata/{ => testing}/seq-scope.md | 0 docs/assays/metadata/{ => testing}/seqfish.md | 0 docs/assays/metadata/{ => testing}/simple.md | 0 docs/assays/metadata/{ => testing}/sims.md | 0 .../metadata/{ => testing}/slide-seq.md | 0 .../metadata/{ => testing}/snareseq2.md | 0 docs/assays/metadata/{ => testing}/starmap.md | 0 .../thicksectionmultiphotonmxif.md | 0 .../metadata/{ => testing}/visium-hd.md | 0 .../{ => testing}/visiumwithprobes.md | 0 docs/assays/metadata/{ => testing}/wgs.md | 0 .../source/reharmonize-legacy-metadata | 1 - 100 files changed, 4321 insertions(+), 737 deletions(-) create mode 100644 docs/assays/metadata/10XMultiome.md create mode 100644 docs/assays/metadata/AutoFluorescence.md create mode 100644 docs/assays/metadata/COMET.md create mode 100644 docs/assays/metadata/CosMx-Proteomics.md create mode 100644 docs/assays/metadata/CosMx-Transcriptomics.md create mode 100644 docs/assays/metadata/CyCIF.md create mode 100644 docs/assays/metadata/CyTOF.md create mode 100644 docs/assays/metadata/DNA-Methylation.md create mode 100644 docs/assays/metadata/EnhancedSRS.md create mode 100644 docs/assays/metadata/FACS.md create mode 100644 docs/assays/metadata/GeoMx.md create mode 100644 docs/assays/metadata/HiFi.md create mode 100644 docs/assays/metadata/IMC.md create mode 100644 docs/assays/metadata/Illumina-Spatial.md create mode 100644 docs/assays/metadata/LightSheet.md create mode 100644 docs/assays/metadata/MERFISH.md create mode 100644 docs/assays/metadata/MPLEx.md create mode 100644 docs/assays/metadata/Olink.md create mode 100644 docs/assays/metadata/PhenoCycler.md create mode 100644 docs/assays/metadata/Pixel-seqV2.md create mode 100644 docs/assays/metadata/RNAseqWithProbes.md create mode 100644 docs/assays/metadata/Raman-Imaging.md create mode 100644 docs/assays/metadata/SIMS.md create mode 100644 docs/assays/metadata/STARmap.md create mode 100644 docs/assays/metadata/SecondHarmonicGeneration.md create mode 100644 docs/assays/metadata/Seq-Scope.md create mode 100644 docs/assays/metadata/Slide-seq.md create mode 100644 docs/assays/metadata/SnareSeq2.md create mode 100644 docs/assays/metadata/ThickSectionMultiphotonMxIF.md create mode 100644 docs/assays/metadata/Visium-HD.md create mode 100644 docs/assays/metadata/VisiumNoProbes.md create mode 100644 docs/assays/metadata/VisiumWithProbes.md create mode 100644 docs/assays/metadata/WGS.md create mode 100644 docs/assays/metadata/iCLAP.md create mode 100644 docs/assays/metadata/link2.png create mode 100644 docs/assays/metadata/seqFISH.md rename docs/assays/{ => metadata/testing}/.directory (88%) rename docs/assays/metadata/{ => testing}/10X-Multiome.md (100%) create mode 100644 docs/assays/metadata/testing/4i.md create mode 100644 docs/assays/metadata/testing/ATACseq.md rename docs/assays/metadata/{ => testing}/Auto-fluorescence.md (100%) create mode 100644 docs/assays/metadata/testing/CODEX.md rename docs/assays/metadata/{ => testing}/Cell-DIVE.md (100%) create mode 100644 docs/assays/metadata/testing/DESI.md create mode 100644 docs/assays/metadata/testing/Histology.md rename docs/assays/metadata/{ => testing}/IMC-2D.md (100%) create mode 100644 docs/assays/metadata/testing/LC-MS.md rename docs/assays/metadata/{ => testing}/Light-Sheet.md (100%) create mode 100644 docs/assays/metadata/testing/MALDI.md create mode 100644 docs/assays/metadata/testing/MIBI.md rename docs/assays/metadata/{ => testing}/MUSIC-(CEDAR).md (100%) create mode 100644 docs/assays/metadata/testing/MUSIC.md rename docs/assays/metadata/{ => testing}/RNAseq-(with-probes).md (100%) create mode 100644 docs/assays/metadata/testing/RNAseq.md rename docs/assays/metadata/{ => testing}/Visium-(no-probes).md (100%) rename docs/assays/metadata/{ => testing}/comet.md (100%) rename docs/assays/metadata/{ => testing}/cosmx-proteomics.md (100%) rename docs/assays/metadata/{ => testing}/cosmx-transcriptomics.md (100%) rename docs/assays/metadata/{ => testing}/cycif.md (100%) rename docs/assays/metadata/{ => testing}/cytof.md (100%) rename docs/assays/metadata/{ => testing}/dna-methylation.md (100%) rename docs/assays/metadata/{ => testing}/enhancedsrs.md (100%) rename docs/assays/metadata/{ => testing}/facs.md (100%) rename docs/assays/metadata/{ => testing}/geomx.md (100%) rename docs/assays/metadata/{ => testing}/hifi.md (100%) rename docs/assays/metadata/{ => testing}/iclap.md (100%) rename docs/assays/metadata/{ => testing}/illumina-spatial.md (100%) rename docs/assays/metadata/{ => testing}/imc.md (100%) create mode 100644 docs/assays/metadata/testing/index.md create mode 100644 docs/assays/metadata/testing/info3.png rename docs/assays/metadata/{ => testing}/merfish.md (100%) rename docs/assays/metadata/{ => testing}/mplex.md (100%) rename docs/assays/metadata/{ => testing}/olink.md (100%) rename docs/assays/metadata/{ => testing}/phenocycler.md (100%) rename docs/assays/metadata/{ => testing}/pixel-seqv2.md (100%) rename docs/assays/metadata/{ => testing}/raman-imaging.md (100%) rename docs/assays/metadata/{ => testing}/secondharmonicgeneration.md (100%) rename docs/assays/metadata/{ => testing}/seq-scope.md (100%) rename docs/assays/metadata/{ => testing}/seqfish.md (100%) rename docs/assays/metadata/{ => testing}/simple.md (100%) rename docs/assays/metadata/{ => testing}/sims.md (100%) rename docs/assays/metadata/{ => testing}/slide-seq.md (100%) rename docs/assays/metadata/{ => testing}/snareseq2.md (100%) rename docs/assays/metadata/{ => testing}/starmap.md (100%) rename docs/assays/metadata/{ => testing}/thicksectionmultiphotonmxif.md (100%) rename docs/assays/metadata/{ => testing}/visium-hd.md (100%) rename docs/assays/metadata/{ => testing}/visiumwithprobes.md (100%) rename docs/assays/metadata/{ => testing}/wgs.md (100%) delete mode 160000 scripts/newMeta2/source/reharmonize-legacy-metadata diff --git a/docs/assays/metadata/10XMultiome.md b/docs/assays/metadata/10XMultiome.md new file mode 100644 index 00000000..a1ab8b90 --- /dev/null +++ b/docs/assays/metadata/10XMultiome.md @@ -0,0 +1,24 @@ +--- +layout: page +--- +# 10X-Multiome + +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|----------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | True | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| True | +| number_of_pre-amplification_pcr_cycles | Numeric | The number of PCR cycles run after the Chromium Controller step and prior to separating the suspension and initiating library construction | | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | True | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/4i.md b/docs/assays/metadata/4i.md index ef03680f..fe7bf3c1 100644 --- a/docs/assays/metadata/4i.md +++ b/docs/assays/metadata/4i.md @@ -1,21 +1,37 @@ +--- +layout: page --- -layout: page-triary ---- - -# 4i Metadata Attributes +# 4i (Iterative Indirect Immunofluorescence Imaging) -Fields that are collected for 4i data, available at ```dataset.metadata.``` -  +
Version 2 (current) -* indicates a required field +## Version 2 (current) -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| source_storage_duration_value * | | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | || time_since_acquisition_instrument_calibration_value | | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | -| contributors_path * | | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | || data_path * | | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | || number_of_antibodies * | | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | || number_of_biomarker_imaging_rounds * | | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | || number_of_total_imaging_rounds * | | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | || slide_id * | | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | || dataset_type * | | The specific type of dataset being produced. Example: RNAseq | ```Visium HD``` ```4i``` ```LC-MS``` ```Thick section Multiphoton MxIF``` ```Light Sheet``` ```ATACseq``` ```Resolve``` ```HiFi-Slide``` ```COMET``` ```MPLEx``` ```10X Multiome``` ```MALDI``` ```Histology``` ```Cell DIVE``` ```FACS``` ```MS Lipidomics``` ```Visium (no probes)``` ```MUSIC``` ```RNAseq``` ```GeoMx (NGS)``` ```GeoMx (nCounter)``` ```RNAseq (with probes)``` ```Singular Genomics G4X``` ```Molecular Cartography``` ```CosMx Transcriptomics``` ```MERFISH``` ```Pixel-seqV2``` ```2D Imaging Mass Cytometry``` ```Confocal``` ```seqFISH``` ```DART-FISH``` ```MIBI``` ```Olink``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```DESI``` ```Xenium``` ```CyCIF``` ```SNARE-seq2``` ```nanoSPLITS``` ```Stereo-seq``` ```Visium (with probes)``` ```SIMS``` ```Auto-fluorescence``` ```CyTOF``` ```CosMx Proteomics``` ```DBiT-seq``` ```PhenoCycler``` ```CODEX``` ```Second Harmonic Generation (SHG)``` ```Seq-Scope``` || analyte_class * | | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein``` ```Lipid + metabolite``` ```Collagen``` ```RNA``` ```Fluorochrome``` ```DNA``` ```Metabolite``` ```DNA + RNA``` ```Saturated lipid``` ```Lipid``` ```Peptide``` ```Protein``` ```Unsaturated lipid``` ```Endogenous fluorophore``` ```Chromatin``` ```Polysaccharide``` || acquisition_instrument_vendor * | | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics``` ```Cytek Biosciences``` ```Thermo Fisher Scientific``` ```Sciex``` ```Vizgen``` ```Leica Microsystems``` ```Akoya Biosciences``` ```Keyence``` ```Andor``` ```Standard BioTools (Fluidigm)``` ```Leica Biosystems``` ```Zeiss Microscopy``` ```Ionpath``` ```Motic``` ```In-House``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Element Biosciences``` ```Hamamatsu``` ```Bruker``` ```Illumina``` ```3DHISTECH``` ```Singular Genomics``` ```Huron Digital Pathology``` ```Resolve Biosciences``` ```NanoString``` ```Cytiva``` ```10x Genomics``` ```Microscopes International``` ```BGI Genomics``` || acquisition_instrument_model * | | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X``` ```NovaSeq X Plus``` ```Cytek Northern Lights``` ```Lightsheet 7``` ```Resolve Biosciences Molecular Cartography``` ```timsTOF HT``` ```timsTOF Pro 2``` ```timsTOF Pro``` ```timsTOF Ultra``` ```timsTOF Ultra 2``` ```timsTOF SCP``` ```Axio Scan.Z1``` ```MALDI timsTOF Flex Prototype``` ```CosMx Spatial Molecular Imager``` ```Unknown``` ```MERSCOPE Ultra``` ```Juno System``` ```timsTOF FleX``` ```Custom: Multiphoton``` ```CyTOF XT``` ```Helios``` ```EVOS M7000``` ```Aperio AT2``` ```Phenocycler-Fusion 2.0``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Observer 3``` ```NanoZoomer-SQ``` ```NanoZoomer S210``` ```NanoZoomer S60``` ```NanoZoomer S360``` ```DM6 B``` ```MoticEasyScan One``` ```In-House``` ```NextSeq 500``` ```BZ-X710``` ```QTRAP 5500``` ```NextSeq 550``` ```HiSeq 2500``` ```HiSeq 4000``` ```NovaSeq 6000``` ```Q Exactive HF``` ```Orbitrap Fusion Lumos Tribrid``` ```Q Exactive``` ```VS200 Slide Scanner``` ```Not applicable``` ```Orbitrap Eclipse Tribrid``` ```MIBIscope``` ```IN Cell Analyzer 2200``` ```timsTOF FleX MALDI-2``` || source_storage_duration_unit * | | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour``` ```month``` ```day``` ```minute``` ```year``` || time_since_acquisition_instrument_calibration_unit | | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month``` ```day``` ```year``` | -| metadata_schema_id * | | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | || preparation_protocol_doi * | | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | || is_targeted * | | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | || antibodies_path * | | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | || parent_sample_id * | | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | || non_global_files | | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | -| cell_boundary_marker_or_stain | | The name of the marker or stain used to identify all cell boundaries in the tissue. This name must exactly match the antibody-targeted molecule marker or non-antibody targeted molecule stain as found in the imaging data. For example, in the case of using the PhenoCycler, ensure the name corresponds to the value in the XPD output file. If multiple markers or stains are employed, list them in a comma-separated format. Example: Pan-Cytokeratin, E-Cadherin | | -| nuclear_marker_or_stain | | The nuclear marker or stain used, which can be an antibody-targeted molecule present in or around the cell nucleus. For protein targets, use the protein or gene symbol that identifies the antibody target, ensuring it matches the antibody target from the panel used or custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets, provide the stain name (e.g., DAPI) and, when applicable, include the associated staining kit and vendor. For the PhenoCycler, ensure the symbol matches the value found in the XPD output file. Example: DAPI | | -| number_of_channels * | | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| number_of_antibodies | Numeric | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | | True | +| number_of_biomarker_imaging_rounds | Numeric | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | | True | +| number_of_total_imaging_rounds | Numeric | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | | True | +| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| antibodies_path | Textfield | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | +| cell_boundary_marker_or_stain | Textfield | The name of the marker or stain used to identify all cell boundaries in the tissue. This name must exactly match the antibody-targeted molecule marker or non-antibody targeted molecule stain as found in the imaging data. For example, in the case of using the PhenoCycler, ensure the name corresponds to the value in the XPD output file. If multiple markers or stains are employed, list them in a comma-separated format. Example: Pan-Cytokeratin, E-Cadherin | | False | +| nuclear_marker_or_stain | Textfield | The nuclear marker or stain used, which can be an antibody-targeted molecule present in or around the cell nucleus. For protein targets, use the protein or gene symbol that identifies the antibody target, ensuring it matches the antibody target from the panel used or custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets, provide the stain name (e.g., DAPI) and, when applicable, include the associated staining kit and vendor. For the PhenoCycler, ensure the symbol matches the value found in the XPD output file. Example: DAPI | | False | +| number_of_channels | Numeric | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | True | +
\ No newline at end of file diff --git a/docs/assays/metadata/ATACseq.md b/docs/assays/metadata/ATACseq.md index f83141cb..3e030c10 100644 --- a/docs/assays/metadata/ATACseq.md +++ b/docs/assays/metadata/ATACseq.md @@ -1,94 +1,257 @@ ---- -layout: page-triary ---- - -# ATACseq Metadata Attributes - -Fields that are collected for ATACseq data, available at ```dataset.metadata.``` -  - -* indicates a required field - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | -| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | -| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | -| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | -| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | -| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | -| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | -| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | -| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | -| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | -| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | -| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | -| barcode_offset *| | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | -| barcode_read *| | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | -| barcode_size *| | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | -| umi_offset *| | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | -| umi_read *| | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | -| umi_size *| | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | -| assay_input_entity *| | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | -| number_of_input_cells_or_nuclei | | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | -| library_adapter_sequence *| | Adapter sequence to be used for adapter trimming | | -| library_average_fragment_size *| | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | -| library_input_amount_value | | The amount of cDNA, after amplification, that was used for library construction. | | -| library_input_amount_unit | | unit of library input amount value | ```ng``` ```ul``` | -| library_output_amount_value | | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | -| library_output_amount_unit | | Units of library final yield. | ```ng``` ```ul``` | -| library_concentration_value *| | The concentration value of the pooled library samples submitted for sequencing. | | -| library_concentration_unit *| | Unit of library_concentration_value | ```ng/ul``` ```nM``` | -| library_layout *| | State whether the library was generated for single-end or paired end sequencing. | ```paired-end``` ```single-end``` | -| number_of_pcr_cycles_for_indexing *| | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | -| library_preparation_kit *| | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```1 slides``` ```4 reactions; PN 1000338``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 reactions; PN 1000187``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | -| sample_indexing_kit *| | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001``` | -| sample_indexing_set *| | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | -| is_technical_replicate *| | Is this a sequencing replicate? | | -| expected_entity_capture_count | | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | -| sequencing_reagent_kit *| | Reagent kit used for sequencing | ```Custom``` ```Illumina``` ```HiSeq 3000/4000 PE Cluster Kit PE-410-1001``` ```PN 1000283, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles)``` ```PN 20046811, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles)``` ```PN 20046812, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles)``` ```PN 20046813, Illumina``` ```NextSeq 2000 P3 Reagent Kit (300 Cycles)``` ```PN 20040561, Illumina``` ```NextSeq 2000 P3 Reagents Kit (100 Cycles)``` ```PN 20040559, Illumina``` ```NextSeq 500/550 Hi Output Kit 150 Cycles``` ```v2.5``` ```PN 20024907, Illumina``` ```NextSeq 500/550 Hi Output Kit 75 Cycles v2.5``` ```PN 20024906, Illumina``` ```NextSeq 500/550 Mid Output Kit 150 Cycles v2.5``` ```PN 20024904, Illumina``` ```NovaSeq 6000 S1 Reagent Kit (200 Cycles)``` ```PN 20012864, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028319, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028318, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles)``` ```PN 20028317, Illumina``` ```NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028316, Illumina``` ```NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles)``` ```PN 20028312, Illumina``` ```NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028313, Illumina``` ```NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles)``` ```PN 20028401, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (100 Cycle)``` ```PN 20104703, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (200 Cycle)``` ```PN 20104704, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (300 Cycle)``` ```PN 20104705, Illumina``` ```NovaSeq X Series 10B Reagent Kit (100 Cycle)``` ```PN 20085596, Illumina``` ```NovaSeq X Series 10B Reagent Kit (200 Cycle)``` ```PN 20085595, Illumina``` ```NovaSeq X Series 10B Reagent Kit (300 Cycle)``` ```PN 20085594``` | -| sequencing_read_format *| | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | -| transposition_reagent_kit | | | | -| transposition_method *| | Modality of capturing accessible chromatin molecules. The kit used, for example. | ```bulkATACseq``` ```sciATACseq``` ```Custom``` ```scATACseq``` | -| sequencing_batch_id | | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | -| capture_batch_id | | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | -| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | -| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | -| preparation_instrument_kit | | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | -| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | - - -  - -## Deprecated Attributes - - These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. - - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | -| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | -| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | -| library_construction_protocols_io_doi | | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | -| library_creation_date | | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | -| library_id | | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | -| operator | | Name of the person responsible for executing the assay. | | -| operator_email | | Email address for the operator. | | -| pi | | Name of the principal investigator responsible for the data. | | -| pi_email | | Email address for the principal investigator. | | -| sample_quality_metric | | This is a quality metric by visual inspection. This should answer the question: Are the nuclei intact and are the nuclei free of significant amounts of debris? This can be captured at a high level, “OK” or “not OK”. | | -| library_pcr_cycles | | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | -| bulk_atac_cell_isolation_protocols_io_doi | | Link to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | -| sc_isolation_enrichment | | The method by which specific cell populations are sorted or enriched. | ```none``` ```FACS``` | -| sc_isolation_protocols_io_doi | | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | -| sc_isolation_quality_metric | | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. "OK" or "not OK", or with more specificity such as "debris", "clump", "low clump". | | -| sc_isolation_tissue_dissociation | | The method by which tissues are dissociated into single cells in suspension. | | -| sc_isolation_cell_number | | Total number of cell/nuclei yielded post dissociation and enrichment. | | -| sequencing_phix_percent | | Percent PhiX loaded to the run | | -| sequencing_read_percent_q30 | | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | -| transposition_transposase_source | | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | ```10X snATAC``` ```In-house``` ```Nextera``` ```10X multiome``` | -| version | | Version of the schema to use when validating this metadata. | ```1``` | -| description | | Free-text description of this assay. | | +--- +layout: page +--- +# ATACseq + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + + +
Version 3 (Latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | True | +| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | +| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | +| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | +| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | +| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | +| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | False | +| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | False | +| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | +| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | +| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | +| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | +| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | +| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | +| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | +| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | +| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | +| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | +| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina``` ```HiSeq 3000/4000 PE Cluster Kit PE-410-1001``` ```PN 1000283, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles)``` ```PN 20046811, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles)``` ```PN 20046812, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles)``` ```PN 20046813, Illumina``` ```NextSeq 2000 P3 Reagent Kit (300 Cycles)``` ```PN 20040561, Illumina``` ```NextSeq 2000 P3 Reagents Kit (100 Cycles)``` ```PN 20040559, Illumina``` ```NextSeq 500/550 Hi Output Kit 150 Cycles``` ```v2.5``` ```PN 20024907, Illumina``` ```NextSeq 500/550 Hi Output Kit 75 Cycles v2.5``` ```PN 20024906, Illumina``` ```NextSeq 500/550 Mid Output Kit 150 Cycles v2.5``` ```PN 20024904, Illumina``` ```NovaSeq 6000 S1 Reagent Kit (200 Cycles)``` ```PN 20012864, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028319, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028318, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles)``` ```PN 20028317, Illumina``` ```NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028316, Illumina``` ```NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles)``` ```PN 20028312, Illumina``` ```NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028313, Illumina``` ```NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles)``` ```PN 20028401, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (100 Cycle)``` ```PN 20104703, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (200 Cycle)``` ```PN 20104704, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (300 Cycle)``` ```PN 20104705, Illumina``` ```NovaSeq X Series 10B Reagent Kit (100 Cycle)``` ```PN 20085596, Illumina``` ```NovaSeq X Series 10B Reagent Kit (200 Cycle)``` ```PN 20085595, Illumina``` ```NovaSeq X Series 10B Reagent Kit (300 Cycle) ``` ```PN 20085594``` | True | +| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | +| transposition_method | Allowable Value | Modality of capturing accessible chromatin molecules. For example, this would be the type of kit that was used. | ```bulkATACseq``` ```sciATACseq``` ```Custom``` ```scATACseq``` | True | +| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | +| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | True | +| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | True | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | + +
+ +
SNARE-seq2 / sciATACseq / snATACseq Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['SNARE-seq2', 'sciATACseq', 'snATACseq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | +| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| is_technical_replicate | boolean | If TRUE, fastq files in dataset need to be merged. | | True | +| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | +| sc_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | +| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol. | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | +| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | +| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | +| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. "OK" or "not OK", or with more specificity such as "debris", "clump", "low clump". | | True | +| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment. | | True | +| transposition_input | Numeric | Number of cell/nuclei input to the assay. | | True | +| transposition_method | Allowable Value | Modality of capturing accessible chromatin molecules. | ['SNARE-Seq2-AC', 'bulkATACseq', 'snATACseq', 'sciATACseq'] | True | +| transposition_transposase_source | Allowable Value | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | ['10X snATAC', 'In-house', 'Nextera', '10X multiome'] | True | +| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". DOI for protocols.io referring to the protocol for this assay. | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming. | | True | +| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). This field is not required for barcoding by single-cell combinatorial indexing. | | False | +| cell_barcode_offset | Textfield | Positions in the read at which the cell barcodes start. Cell barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. (Does not apply to sciATACseq, SNARE-seq and BulkATAC.) | | False | +| cell_barcode_size | Textfield | Length of the cell barcode in base pairs. Cell barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. (Does not apply to sciATACseq, SNARE-seq and BulkATAC.) | | False | +| library_pcr_cycles | Numeric | Number of PCR cycles to enrich for accessible chromatin fragments. | | True | +| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library generation (figure in Descriptions section) | | True | +| library_final_yield | Numeric | Total ng of library after final pcr amplification step. | | True | +| library_final_yield_unit | Allowable Value | Units for library_final_yield | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
SNARE-seq2 / sciATACseq / snATACseq Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | +| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| sc_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | +| sc_isolation_entity | Textfield | The type of single cell entity derived from isolation protocol | | True | +| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | +| sc_isolation_enrichment | Textfield | The method by which specific cell populations are sorted or enriched. | | False | +| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | +| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | +| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | +| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | +| is_technical_replicate | boolean | Is the sequencing reaction run in repliucate, TRUE or FALSE | | True | +| cell_barcode_read | Textfield | Which read file contains the cell barcode | | True | +| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | True | +| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | True | +| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | +| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
bulkATACseq Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['bulkATACseq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | +| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| bulk_transposition_input_number_nuclei | Textfield | A number (no comma separators) | | True | +| bulk_atac_cell_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | True | +| is_technical_replicate | boolean | Is this a sequencing replicate? | | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | +| library_concentration_unit | Allowable Value | Unit of library_concentration_value | ['nM'] | False | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_creation_date | Datetime | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | False | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_pcr_cycles | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | True | +| library_preparation_kit | Textfield | Reagent kit used for library preparation | | True | +| sample_quality_metric | Textfield | This is a quality metric by visual inspection. This should answerthe question: Are the nuclei intact and are the nuclei free of significant amountsof debris? This can be captured at a high level, “OK” or “notOK”. | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | +| transposition_method | Textfield | Modality of capturing accessible chromatin molecules. The kit used, for example. | | True | +| transposition_transposase_source | Textfield | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ + + +
bulkATACseq 0 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['bulkATACseq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | +| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| bulk_transposition_input_number_nuclei | Textfield | A number (no comma separators) | | True | +| bulk_atac_cell_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | True | +| is_technical_replicate | boolean | Is this a sequencing replicate? | | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | +| library_concentration_unit | Allowable Value | Unit of library_concentration_value | ['nM'] | False | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_creation_date | Datetime | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | False | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_pcr_cycles | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | True | +| library_preparation_kit | Textfield | Reagent kit used for library preparation | | True | +| sample_quality_metric | Textfield | This is a quality metric by visual inspection. This should answerthe question: Are the nuclei intact and are the nuclei free of significant amountsof debris? This can be captured at a high level, “OK” or “notOK”. | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | +| transposition_method | Textfield | Modality of capturing accessible chromatin molecules. The kit used, for example. | | True | +| transposition_transposase_source | Textfield | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
diff --git a/docs/assays/metadata/AutoFluorescence.md b/docs/assays/metadata/AutoFluorescence.md new file mode 100644 index 00000000..6abca683 --- /dev/null +++ b/docs/assays/metadata/AutoFluorescence.md @@ -0,0 +1,105 @@ +--- +layout: page +--- +# Auto-fluorescence + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
Version 2 (Latest) + +## Version 2 (Latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | +| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | +| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement | ```month``` ```day``` ```year``` | False | +| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
+ +
Version 1 + +## Version 1 + +| Attribute | Type | Description | AllowableValues | Required | +|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['AF'] | True | +| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | False | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| resolution_x_value | Numeric | The width of a pixel. | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that this is resolution within a given sample: z-pitch (resolution_z_value) is the increment distance between image slices, ie. the microscope stage is moved up or down in increments to capture images of several focal planes. | | True | +| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | +| number_of_channels | Numeric | Number of channels capturing the emission spectrum from natural fluorophores in the sample. | | True | +| overall_protocols_io_doi | Textfield | DOI for protocols.io referring to the overall protocol for the assay. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
Version 0 + +## Version 0 + +| Attribute | Type | Description | AllowableValues | Required | +|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['AF'] | True | +| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | False | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| resolution_x_value | Numeric | The width of a pixel. | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that this is resolution within a given sample: z-pitch (resolution_z_value) is the increment distance between image slices, ie. the microscope stage is moved up or down in increments to capture images of several focal planes. | | True | +| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | +| number_of_channels | Numeric | Number of channels capturing the emission spectrum from natural fluorophores in the sample. | | True | +| overall_protocols_io_doi | Textfield | DOI for protocols.io referring to the overall protocol for the assay. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
diff --git a/docs/assays/metadata/CODEX.md b/docs/assays/metadata/CODEX.md index 4550574c..f2efa40f 100644 --- a/docs/assays/metadata/CODEX.md +++ b/docs/assays/metadata/CODEX.md @@ -1,62 +1,120 @@ ---- -layout: page-triary ---- - -# CODEX Metadata Attributes - -Fields that are collected for CODEX data, available at ```dataset.metadata.``` -  - -* indicates a required field - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | -| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | -| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | -| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | -| acquisition_instrument_vendor *| | An acquisition_instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | -| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | -| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | -| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | -| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | -| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | -| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | -| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | -| antibodies_path *| | Relative path to file with antibody information for this dataset. | | -| preparation_instrument_vendor *| | The manufacturer of the instrument used to prepare the sample for the assay. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | -| preparation_instrument_model *| | The model number/name of the instrument used to prepare the sample for the assay | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | -| total_run_time_value | | How long the tissue was on the acquisition instrument. | | -| total_run_time_unit | | The units for the total run time unit field. | ```Hour``` ```Minute``` | -| number_of_antibodies *| | Number of antibodies | | -| number_of_channels *| | Number of fluorescent channels imaged during each cycle. | | -| number_of_biomarker_imaging_rounds *| | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo application, 2. fluor application, 3. washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | -| number_of_total_imaging_rounds *| | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | -| slide_id | | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | -| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | - - -  - -## Deprecated Attributes - - These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. - - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | -| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | -| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | -| operator | | Name of the person responsible for executing the assay. | | -| operator_email | | Email address for the operator. | | -| pi | | Name of the principal investigator responsible for the data. | | -| pi_email | | Email address for the principal investigator. | | -| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | -| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | -| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | -| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | -| resolution_z_unit | | The unit of incremental distance between image slices. | ```mm``` ```um``` ```nm``` | -| resolution_z_value | | Optional if assay does not have multiple z-levels. Note that this is resolution within a given sample: z-pitch (resolution_z_value) is the increment distance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stage is moved up or down in increments of 1.5um to capture images of several focal planes. The best one will be used & the rest discarded. The thickness of the sample itself is sample metadata. | | +--- +layout: page +--- +# CODEX + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
Version 2 (Latest) + +## Version 2 + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | +| number_of_biomarker_imaging_rounds | Numeric | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo application, 2. fluor application, 3. washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | True | +| number_of_total_imaging_rounds | Numeric | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| total_run_time_value | Numeric | How long the tissue was on the acquisition instrument. | | False | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| total_run_time_unit | Allowable Value | The units for the total run time unit field. | ```Hour``` ```Minute``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | True | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
+ +
Version 1 + +## Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['CODEX', 'CODEX2'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes', 'No'] | True | +| acquisition_instrument_vendor | Allowable Value | An acquisition_instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing molecular mass. | ['Keyence', 'Zeiss'] | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | ['BZ-X800', 'BZ-X710', 'Axio Observer Z1'] | True | +| resolution_x_value | Numeric | The width of a pixel. (Akoya pixel is 377nm square) | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of width of a pixel.(nm) | ['mm', 'um', 'nm'] | False | +| resolution_y_value | Numeric | The height of a pixel. (Akoya pixel is 377nm square) | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of height of a pixel. (nm) | ['mm', 'um', 'nm'] | False | +| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that thisis resolution within a given sample: z-pitch (resolution_z_value) is the incrementdistance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stageis moved up or down in increments of 1.5um to capture images of several focalplanes. The best one will be used & the rest discarded. The thickness of the sampleitself is sample metadata. | | False | +| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare the sample for theassay. | ['CODEX'] | True | +| preparation_instrument_model | Allowable Value | The model number/name of the instrument used to prepare the samplefor the assay | ['version 1 robot', 'prototype robot - Stanford/Nolan Lab'] | True | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | +| number_of_cycles | Numeric | Number of cycles of 1. oligo application, 2. fluor application, 3.washes | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | +| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagentsfor the assay. | | True | +| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | + +
+ +
Version 0 + +## Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['CODEX'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes', 'No'] | True | +| acquisition_instrument_vendor | Allowable Value | An acquisition_instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing molecular mass. | ['Keyence', 'Zeiss'] | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | ['BZ-X800', 'BZ-X710', 'Axio Observer Z1'] | True | +| resolution_x_value | Numeric | The width of a pixel. (Akoya pixel is 377nm square) | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of width of a pixel.(nm) | ['mm', 'um', 'nm'] | False | +| resolution_y_value | Numeric | The height of a pixel. (Akoya pixel is 377nm square) | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of height of a pixel. (nm) | ['mm', 'um', 'nm'] | False | +| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that thisis resolution within a given sample: z-pitch (resolution_z_value) is the incrementdistance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stageis moved up or down in increments of 1.5um to capture images of several focalplanes. The best one will be used & the rest discarded. The thickness of the sampleitself is sample metadata. | | False | +| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | +| | Textfield | | | | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare the sample for theassay. | ['CODEX'] | True | +| preparation_instrument_model | Allowable Value | The model number/name of the instrument used to prepare the samplefor the assay | ['version 1 robot', 'prototype robot - Stanford/Nolan Lab'] | True | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | +| number_of_cycles | Numeric | Number of cycles of 1. oligo application, 2. fluor application, 3.washes | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | +| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagentsfor the assay. | | True | +| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | + +
diff --git a/docs/assays/metadata/COMET.md b/docs/assays/metadata/COMET.md new file mode 100644 index 00000000..7a7348b1 --- /dev/null +++ b/docs/assays/metadata/COMET.md @@ -0,0 +1,37 @@ +--- +layout: page +--- +# COMET + +
Version 2.0 (use this one) + +## Version 2.0 (use this one) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| number_of_antibodies | Numeric | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | | True | +| number_of_biomarker_imaging_rounds | Numeric | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | | True | +| number_of_total_imaging_rounds | Numeric | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | | True | +| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq``` | True | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| antibodies_path | Textfield | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | +| cell_boundary_marker_or_stain | Textfield | The name of the marker or stain used to identify all cell boundaries in the tissue. This name must exactly match the antibody-targeted molecule marker or non-antibody targeted molecule stain as found in the imaging data. For example, in the case of using the PhenoCycler, ensure the name corresponds to the value in the XPD output file. If multiple markers or stains are employed, list them in a comma-separated format. Example: Pan-Cytokeratin, E-Cadherin | | False | +| nuclear_marker_or_stain | Textfield | The nuclear marker or stain used, which can be an antibody-targeted molecule present in or around the cell nucleus. For protein targets, use the protein or gene symbol that identifies the antibody target, ensuring it matches the antibody target from the panel used or custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets, provide the stain name (e.g., DAPI) and, when applicable, include the associated staining kit and vendor. For the PhenoCycler, ensure the symbol matches the value found in the XPD output file. Example: DAPI | | False | +| number_of_channels | Numeric | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/CosMx-Proteomics.md b/docs/assays/metadata/CosMx-Proteomics.md new file mode 100644 index 00000000..d50cdc02 --- /dev/null +++ b/docs/assays/metadata/CosMx-Proteomics.md @@ -0,0 +1,43 @@ +--- +layout: page +--- +# CosMx Proteomics + +
Version 2 (current) + +## Version 2 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | False | +| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | False | +| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | +| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | +| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | +| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | +| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | +| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | +| oligo_probe_panel | Assigned Value | The oligo probe panel used to target genes and/or proteins. If there is a core panel along with add-on modules, the core panel should be selected in this field. Any additional panels utilized should be documented in the "additional_panels_used.csv" file, which must be uploaded alongside the dataset. Example: 10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363 | ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-MsWTA-4```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (RNA, 1000 Plex); PN CMX-M-NEUP-R```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 16 rxns; PN 1000420```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 4 rxns x 4 BC; PN 1000475```, ```NanoString Technologies; CosMx Human 6K Discovery Panel (RNA, 6175 Plex); PN 121500041```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome, 4 rxns x 1 BC; PN 1000474```, ```10x Genomics; Xenium Human Multi-Tissue and Cancer Panel v1; PN 1000626```, ```NanoString Technologies; CosMx Human Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-H-USCP-1KP-R```, ```10x Genomics; Xenium Custom Gene Expression Panel (up to 50 genes); PN 1000464```, ```NanoString Technologies; CosMx Hs Univ Cell (RNA, 1000 Plex); PN 121500002```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365```, ```10x Genomics; Xenium Mouse Multi-Tissue Atlassing Panel; PN 1000627```, ```10x Genomics; Xenium Custom Gene Expression Panel (51-100 genes); PN 1000561```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 64 rxns; PN 1000456```, ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-HuWTA-4```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit v2.0 - Small; PN 1000667```, ```NanoString Technologies; CosMx Mouse Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-M-USCP-1KP-R```, ```NanoString Technologies; CosMx Human Immuno-Oncology Panel (Protein, 64 Plex); PN CMX-H-IOP-64P-P```, ```Custom```, ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466```, ```10x Genomics; Xenium Human Colon Gene Expression Panel; PN 1000642```, ```10x Genomics; Chromium Next GEM Single Cell Fixed RNA Mouse Transcriptome Probe Kit, 64 rxns; PN 1000492```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (Protein, 64 Plex); PN CMX-M-Neuro-64P-P```, ```NanoString Technologies; CosMx Hs WTX RNA Panel Kit, 2 slides: PN 121500047```, ```10x Genomics; Xenium Human Lung Gene Expression Panel; PN 1000601```, ```10x Genomics; Xenium Prime 5K Human Pan Tissue & Pathways Panel; PN 1000724``` | False | +| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | +| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | +| roi_label | Textfield | The label for the region of interest (ROI). For Resolve and CosMx, this corresponds to the field of view (FOV) label. In the case of Xenium, it refers to the ID of the region containing the analysis. For GeoMx, this information can be located in the "Initial Dataset" spreadsheet, which can be downloaded from within the Data Analysis Suite. Example: Decidua | | False | +| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | +| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | + +
\ No newline at end of file diff --git a/docs/assays/metadata/CosMx-Transcriptomics.md b/docs/assays/metadata/CosMx-Transcriptomics.md new file mode 100644 index 00000000..62695999 --- /dev/null +++ b/docs/assays/metadata/CosMx-Transcriptomics.md @@ -0,0 +1,88 @@ +--- +layout: page +--- +# CosMx Transcriptomics + +
Version 3 (current) + +## Version 3 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial v0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics``` | True | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | False | +| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | False | +| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | +| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | +| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | +| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | +| proteinasek_concentration | Numeric | The concentration of the enzyme Proteinase K within a sample, measured in micrograms per milliliter (ug/ml). Example: 10 | | False | +| proteinasek_incubation_time_value | Numeric | The duration for which a sample is incubated with Proteinase K. Example: 15 | | False | +| proteinasek_incubation_time_unit | Assigned Value | The unit of measurement for the proteinaseK incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | +| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | +| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | +| oligo_probe_panel | Assigned Value | The oligo probe panel used to target genes and/or proteins. If there is a core panel along with add-on modules, the core panel should be selected in this field. Any additional panels utilized should be documented in the "additional_panels_used.csv" file, which must be uploaded alongside the dataset. Example: 10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363 | ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-MsWTA-4```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 16 rxns x 16 BC; PN 1000547```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (RNA, 1000 Plex); PN CMX-M-NEUP-R```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 16 rxns; PN 1000420```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 4 rxns x 4 BC; PN 1000475```, ```NanoString Technologies; CosMx Human 6K Discovery Panel (RNA, 6175 Plex); PN 121500041```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome, 4 rxns x 1 BC; PN 1000474```, ```10x Genomics; Xenium Human Multi-Tissue and Cancer Panel v1; PN 1000626```, ```NanoString Technologies; CosMx Human Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-H-USCP-1KP-R```, ```10x Genomics; GEM-X Flex Human Transcriptome Probe Kit, 16 samples; PN 1000785```, ```10x Genomics; Xenium Custom Gene Expression Panel (up to 50 genes); PN 1000464```, ```NanoString Technologies; CosMx Hs Univ Cell (RNA, 1000 Plex); PN 121500002```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365```, ```10x Genomics; Xenium Mouse Multi-Tissue Atlassing Panel; PN 1000627```, ```10x Genomics; Xenium Custom Gene Expression Panel (51-100 genes); PN 1000561```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 64 rxns; PN 1000456```, ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-HuWTA-4```, ```NanoString Technologies; GeoMx Human IO Proteome Atlas, 4 slides; PN 121300160```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit v2.0 - Small; PN 1000667```, ```NanoString Technologies; CosMx Mouse Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-M-USCP-1KP-R```, ```NanoString Technologies; CosMx Human Immuno-Oncology Panel (Protein, 64 Plex); PN CMX-H-IOP-64P-P```, ```Custom```, ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466```, ```10x Genomics; Xenium Human Colon Gene Expression Panel; PN 1000642```, ```10x Genomics; Chromium Next GEM Single Cell Fixed RNA Mouse Transcriptome Probe Kit, 64 rxns; PN 1000492```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (Protein, 64 Plex); PN CMX-M-Neuro-64P-P```, ```NanoString Technologies; CosMx Hs WTX RNA Panel Kit, 2 slides: PN 121500047```, ```10x Genomics; Xenium Human Lung Gene Expression Panel; PN 1000601```, ```10x Genomics; Xenium Prime 5K Human Pan Tissue & Pathways Panel; PN 1000724``` | True | +| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | +| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | +| roi_label | Textfield | The label for the region of interest (ROI). For Resolve and CosMx, this corresponds to the field of view (FOV) label. In the case of Xenium, it refers to the ID of the region containing the analysis. For GeoMx, this information can be located in the "Initial Dataset" spreadsheet, which can be downloaded from within the Data Analysis Suite. Example: Decidua | | False | +| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | +| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | + +
+ + +
Version 2 + +## Version 2 + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial v0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics``` | True | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | False | +| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | False | +| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | +| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | +| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | +| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | +| proteinasek_concentration | Numeric | The concentration of the enzyme Proteinase K within a sample, measured in micrograms per milliliter (ug/ml). Example: 10 | | False | +| proteinasek_incubation_time_value | Numeric | The duration for which a sample is incubated with Proteinase K. Example: 15 | | False | +| proteinasek_incubation_time_unit | Assigned Value | The unit of measurement for the proteinaseK incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | +| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | +| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | +| oligo_probe_panel | Assigned Value | The oligo probe panel used to target genes and/or proteins. If there is a core panel along with add-on modules, the core panel should be selected in this field. Any additional panels utilized should be documented in the "additional_panels_used.csv" file, which must be uploaded alongside the dataset. Example: 10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363 | ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-MsWTA-4```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 16 rxns x 16 BC; PN 1000547```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (RNA, 1000 Plex); PN CMX-M-NEUP-R```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 16 rxns; PN 1000420```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 4 rxns x 4 BC; PN 1000475```, ```NanoString Technologies; CosMx Human 6K Discovery Panel (RNA, 6175 Plex); PN 121500041```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome, 4 rxns x 1 BC; PN 1000474```, ```10x Genomics; Xenium Human Multi-Tissue and Cancer Panel v1; PN 1000626```, ```NanoString Technologies; CosMx Human Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-H-USCP-1KP-R```, ```10x Genomics; GEM-X Flex Human Transcriptome Probe Kit, 16 samples; PN 1000785```, ```10x Genomics; Xenium Custom Gene Expression Panel (up to 50 genes); PN 1000464```, ```NanoString Technologies; CosMx Hs Univ Cell (RNA, 1000 Plex); PN 121500002```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365```, ```10x Genomics; Xenium Mouse Multi-Tissue Atlassing Panel; PN 1000627```, ```10x Genomics; Xenium Custom Gene Expression Panel (51-100 genes); PN 1000561```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 64 rxns; PN 1000456```, ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-HuWTA-4```, ```NanoString Technologies; GeoMx Human IO Proteome Atlas, 4 slides; PN 121300160```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit v2.0 - Small; PN 1000667```, ```NanoString Technologies; CosMx Mouse Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-M-USCP-1KP-R```, ```NanoString Technologies; CosMx Human Immuno-Oncology Panel (Protein, 64 Plex); PN CMX-H-IOP-64P-P```, ```Custom```, ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466```, ```10x Genomics; Xenium Human Colon Gene Expression Panel; PN 1000642```, ```10x Genomics; Chromium Next GEM Single Cell Fixed RNA Mouse Transcriptome Probe Kit, 64 rxns; PN 1000492```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (Protein, 64 Plex); PN CMX-M-Neuro-64P-P```, ```NanoString Technologies; CosMx Hs WTX RNA Panel Kit, 2 slides: PN 121500047```, ```10x Genomics; Xenium Human Lung Gene Expression Panel; PN 1000601```, ```10x Genomics; Xenium Prime 5K Human Pan Tissue & Pathways Panel; PN 1000724``` | True | +| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | +| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | +| roi_label | Textfield | The label for the region of interest (ROI). For Resolve and CosMx, this corresponds to the field of view (FOV) label. In the case of Xenium, it refers to the ID of the region containing the analysis. For GeoMx, this information can be located in the "Initial Dataset" spreadsheet, which can be downloaded from within the Data Analysis Suite. Example: Decidua | | False | +| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | +| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/CyCIF.md b/docs/assays/metadata/CyCIF.md new file mode 100644 index 00000000..2d293e15 --- /dev/null +++ b/docs/assays/metadata/CyCIF.md @@ -0,0 +1,34 @@ +--- +layout: page +--- +# CyCIF + +
Version 2 (current) + +## Version 2 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_biomarker_imaging_rounds | Numeric | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo and fluor application, 2. imaging, 3. removal of oligo and fluor along washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | True | +| number_of_total_imaging_rounds | Numeric | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | +| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | +| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | +| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | +| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/CyTOF.md b/docs/assays/metadata/CyTOF.md new file mode 100644 index 00000000..f7f95499 --- /dev/null +++ b/docs/assays/metadata/CyTOF.md @@ -0,0 +1,38 @@ +--- +layout: page +--- +# CyTOF + +
Version 2 (Latest) + +## Version 2 (Latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|----------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------|----------| +| lab_id | Textfield | An internal attribute labs can use to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This attribute will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | +| dataset_type | Textfield | The specific type of dataset being produced. | | True | +| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Textfield | The time duration unit of measurement | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Textfield | The time unit of measurement | | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: [https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1](https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1). | | True | +| is_targeted | Assigned Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Custom``` ```None``` ```Sigma Aldrich; Cisplatin 25mg; PN P4394``` ```Standard BioTools; Cell-ID Cisplatin-198Pt 100 uL; PN 201198``` ```Standard BioTools; Cell-ID Intercalator-103Rh 2,000 um; PN 201103B``` ```Standard BioTools; Cell-ID Cisplatin-196Pt 100 uL; PN 201196``` ```Standard BioTools; Cell-ID Cisplatin 100 uL; PN 201064``` ```Standard BioTools; Cell-ID Intercalator-103Rh 500 um; PN 201103A``` ```Standard BioTools; Cell-ID Cisplatin-194Pt 100 uL; PN 201194``` ```Standard BioTools; Cell-ID Cisplatin-195Pt 100 uL; PN 201195``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| number_of_mass_channels | Numeric | The number of mass channels that measure the expression of markers in single cells. | | False | +| is_erythrocyte_lysis_performed | Assigned Value | Process in which red blood cells (RBCs) are broken down in the sample prior to analysis, thereby allowing researchers to focus primarily on white blood cells (WBCs). | ```Yes``` ```No``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | False | +| antibody_reagent_kit | Textfield | The kit containing the set of antibodies pre-conjugated with different heavy metal isotopes used to simultaneously detect and quantify multiple protein markers on individual cells by attaching these metal-labeled antibodies to specific cellular targets, essentially acting as the key component for labeling cells with the various markers needed for analysis on the CyTOF machine. | | False | +| viability_reagent_kit | Textfield | The kit used to differentiate between live and dead cells within a sample by selectively staining dead cells with a dye that can be detected by the instrument, allowing researchers to exclude dead cell data from their analysis and ensure accurate results when studying cell populations. | | False | +| is_cell_activation_performed | Assigned Value | Process by which ligand is binded to its receptors on a cell, which enhances the cell's ability to respond to various stimuli. | ```Yes``` ```No```e | False | +| activation_stimulus | Textfield | Specific type of stimulus used to provoke cell activation. Examples would include PMA/ionomycin or CD28in/brefeldin A. This field is required if "is_cells_activation performed" is Yes. | | False | +| is_fcr_blocking_applied | Assigned Value | Process by which a reagent has been added to the staining procedure to block the binding of antibodies to Fc receptors (FcRs) on cells, preventing non-specific binding and ensuring that only the intended target antigen is detected by the antibodies; essentially, it helps to minimize false positive signals by preventing antibodies from attaching to the cell via their Fc region instead of the antigen-specific binding site. | ```Yes``` ```No``` | False | +| is_heparin_used | Assigned Value | Indicates whether heparin was used ("Yes") or not ("No") during staining to prevent non-specific binding of metal-labeled antibodies to eosinophils to reduce background noise. |```Yes``` ```No``` | False | +| loaded_cell_concentration_value | Numeric | The number of cells present within a given volume of liquid for the experiment immediately prior to the experiment, essentially indicating how densely packed the cells are in a solution. | | False | +| loaded_cell_concentration_unit | Textfield | Unit of measure for cell concentration, e.g. cells per milliliter (cells/mL). | | False | +| instrument_calibration_bead_kit | Textfield | A set of beads of known mass intensity used to adjust the settings of a flow cytometer to ensure accurate measurements. | | False | +| calibration_kit_lot_number | Textfield | Manufacturer's lot number for the calibration bead kit used for the experiment. | | False | diff --git a/docs/assays/metadata/DESI.md b/docs/assays/metadata/DESI.md index c651b8f3..7ead4ea6 100644 --- a/docs/assays/metadata/DESI.md +++ b/docs/assays/metadata/DESI.md @@ -1,69 +1,43 @@ ---- -layout: page-triary ---- - -# DESI Metadata Attributes - -Fields that are collected for DESI data, available at ```dataset.metadata.``` -  - -* indicates a required field - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | -| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | -| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | -| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | -| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | -| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | -| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | -| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | -| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | -| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | -| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | -| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | -| ms_ionization_technique *| | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20``` | -| ms_scan_mode *| | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | -| mass_analysis_polarity *| | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | -| mass_to_charge_range_low_value | | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | -| mass_to_charge_range_high_value | | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | -| mass_resolving_power *| | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | -| mass_to_charge_resolving_power | | The peak (m/z) used to calculate the resolving power. | | -| ion_mobility | | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```TIMS``` ```SLIM``` ```FAIMS``` ```DTIMS``` ```cIMS``` ```TWIMS``` | -| matrix_deposition_method | | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | -| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | -| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | -| preparation_matrix | | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | -| desorption_solvent *| | Solvent composition for conducting nanospray desorption electrospray ionization (nanoDESI) or desorption electrospray ionization (DESI). | ```Acetonitrile:Dimethylformamide (ACN:DMF)``` ```Acetonitrile:Water (ACN:H2O)``` ```Ethanol:Dimethylformamide (EtOH:DMF)``` ```Ethanol:Water (EtOH:H2O)``` ```Methanol:Ethanol (MeOH:EtOH)``` ```Methanol:Water (MeOH:H2O)``` | -| desorption_solvent_flow_rate_value *| | The rate of flow of the solvent into a spray. | | -| desorption_solvent_flow_rate_unit *| | Units of the rate of solvent flow. | ```nL/min``` ```uL/min``` | -| analysis_protocol_doi *| | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | -| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | - - -  - -## Deprecated Attributes - - These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. - - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | -| description | | Free-text description of this assay. | | -| overall_protocols_io_doi | | DOI for protocols.io referring to the overall protocol for the assay. | | -| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | -| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | -| operator | | Name of the person responsible for executing the assay. | | -| operator_email | | Email address for the operator. | | -| pi | | Name of the principal investigator responsible for the data. | | -| pi_email | | Email address for the principal investigator. | | -| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | -| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | -| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | -| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | -| version | | Version of the schema to use when validating this metadata. | ```1``` | +--- +layout: page +--- +# DESI + +
Version 2 (Latest) + +## Version 2 (Latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | +| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | True | +| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | True | +| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | True | +| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | +| matrix_deposition_method | Allowable Value | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | +| preparation_matrix | Allowable Value | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| ion_mobility | Allowable Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```cIMS``` ```DTIMS``` ```FAIMS``` ```SLIM``` ```TIMS``` ```TWIMS``` | False | +| desorption_solvent | Allowable Value | Solvent composition for conducting nanospray desorption electrospray ionization (nanoDESI) or desorption electrospray ionization (DESI). | ```Acetonitrile:Dimethylformamide (ACN:DMF)``` ```Acetonitrile:Water (ACN:H2O)``` ```Ethanol:Dimethylformamide (EtOH:DMF)``` ```Ethanol:Water (EtOH:H2O)``` ```Methanol:Ethanol (MeOH:EtOH)``` ```Methanol:Water (MeOH:H2O)``` | True | +| desorption_solvent_flow_rate_value | Numeric | The rate of flow of the solvent into a spray. | | True | +| desorption_solvent_flow_rate_unit | Allowable Value | Units of the rate of solvent flow. | ```nL/min``` ```uL/min``` | True | +| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | +| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | +| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
diff --git a/docs/assays/metadata/DNA-Methylation.md b/docs/assays/metadata/DNA-Methylation.md new file mode 100644 index 00000000..f87faab0 --- /dev/null +++ b/docs/assays/metadata/DNA-Methylation.md @@ -0,0 +1,28 @@ +--- +layout: page +--- +# DNA Methylation + +
Version 2 (current) + +## Version 2 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial ver0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```DNA Methylation```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF``` | True | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: d70bfe24-e82a-46cb-9369-28ae03660d97 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/EnhancedSRS.md b/docs/assays/metadata/EnhancedSRS.md new file mode 100644 index 00000000..7a906777 --- /dev/null +++ b/docs/assays/metadata/EnhancedSRS.md @@ -0,0 +1,34 @@ +--- +layout: page +--- +# Enhanced-SRS + +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | +| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | +| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
diff --git a/docs/assays/metadata/FACS.md b/docs/assays/metadata/FACS.md new file mode 100644 index 00000000..8c2d8d98 --- /dev/null +++ b/docs/assays/metadata/FACS.md @@ -0,0 +1,39 @@ +--- +layout: page +--- +# FACS + +
Version 2 (current) + +## Version 2 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | +| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | +| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| is_erythrocyte_lysis_performed | Radio | Process in which red blood cells (RBCs) are broken down in the sample prior to analysis, thereby allowing researchers to focus primarily on white blood cells (WBCs). | ```Yes,No``` | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | False | +| antibody_reagent_kit | Assigned Value | The kit containing the set of antibodies pre-conjugated with different heavy metal isotopes used to simultaneously detect and quantify multiple protein markers on individual cells by attaching these metal-labeled antibodies to specific cellular targets, essentially acting as the key component for labeling cells with the various markers needed for analysis on the CyTOF machine. | ```Standard BioTools; Maxpar Nuclear Antigen Staining Kit; PN 201603```, ```Standard BioTools; Maxpar Phosphoprotein Staining Kit; PN 201604```, ```Standard BioTools; Maxpar Cell Surface Staining Kit; PN 201601```, ```Standard BioTools; Maxpar Cytoplasmic/Secreted Antigen Staining Kit; PN 201602```, ```Custom``` | True | +| viability_reagent_kit | Assigned Value | The kit used to differentiate between live and dead cells within a sample by selectively staining dead cells with a dye that can be detected by the instrument, allowing researchers to exclude dead cell data from their analysis and ensure accurate results when studying cell populations. | ```Sigma Aldrich; Cisplatin 25mg; PN P4394```, ```Standard BioTools; Cell-ID Cisplatin-198Pt 100 uL; PN 201198```, ```None```, ```Standard BioTools; Cell-ID Intercalator-103Rh 2,000 um; PN 201103B```, ```Standard BioTools; Cell-ID Cisplatin-196Pt 100 uL; PN 201196```, ```Standard BioTools; Cell-ID Cisplatin 100 uL; PN 201064```, ```Standard BioTools; Cell-ID Intercalator-103Rh 500 um; PN 201103A```, ```Standard BioTools; Cell-ID Cisplatin-194Pt 100 uL; PN 201194```, ```Standard BioTools; Cell-ID Cisplatin-195Pt 100 uL; PN 201195```, ```Custom``` | True | +| is_cell_activation_performed | Radio | Process by which ligand is binded to its receptors on a cell, which enhances the cell's ability to respond to various stimuli. | ```Yes,No``` | True | +| activation_stimulus | Textfield | Specific type of stimulus used to provoke cell activation. Examples would include PMA/ionomycin or CD28in/brefeldin A. This field is required if "is_cells_activation performed" is Yes. | | False | +| is_fcr_blocking_applied | Radio | Process by which a reagent has been added to the staining procedure to block the binding of antibodies to Fc receptors (FcRs) on cells, preventing non-specific binding and ensuring that only the intended target antigen is detected by the antibodies; essentially, it helps to minimize false positive signals by preventing antibodies from attaching to the cell via their Fc region instead of the antigen-specific binding site. | ```Yes,No``` | True | +| is_heparin_used | Radio | Indicates whether heparin was used ("Yes") or not ("No") during staining to prevent non-specific binding of metal-labeled antibodies to eosinophils to reduce background noise. | ```Yes,No``` | True | +| loaded_cell_concentration_value | Numeric | The number of cells present within a given volume of liquid for the experiment immediately prior to the experiment, essentially indicating how densely packed the cells are in a solution. | | False | +| loaded_cell_concentration_unit | Assigned Value | Unit of measure for cell concentration, e.g. cells per milliliter (cells/mL). | ```cells/mL``` | False | +| instrument_calibration_bead_kit | Assigned Value | A set of beads of known mass intensity used to adjust the settings of a flow cytometer to ensure accurate measurements. | ```Standard BioTools; EQ Six Element Calibration Beads 100 mL; PN 201245```, ```Standard BioTools; EQ Four Element Calibration Beads 100 mL; PN 201078```, ```None```, ```Standard BioTools; CyTOF Calibration Beads; PN 201073```, ```Custom``` | True | +| calibration_kit_lot_number | Textfield | Manufacturer's lot number for the calibration bead kit used for the experiment. | | True | + +
diff --git a/docs/assays/metadata/GeoMx.md b/docs/assays/metadata/GeoMx.md new file mode 100644 index 00000000..0747a560 --- /dev/null +++ b/docs/assays/metadata/GeoMx.md @@ -0,0 +1,97 @@ +--- +layout: page +--- +# GeoMx + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
NGS Version 2 (current) + +## NGS Version 2 (current) + +| attribute | type | description | value | required | +|-----------------------------------------------------|----------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------|------------| +| dataset_type | Textfield | The specific type of dataset being produced. | | True | +| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Textfield | The time duration unit of measurement | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Textfield | The time unit of measurement | | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ['Yes', 'No'] | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | +| mapped_area_unit | Textfield | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | +| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | True | +| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | True | +| target_retrieval_incubation_time_unit | Textfield | The units for target retrieval incubation time value. | | True | +| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | False | +| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | False | +| proteinasek_incubation_time_unit | Textfield | The units for proteinaseK incubation time value. | | False | +| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | True | +| is_roi_segmentation_performed | Allowable Value | Was the image segmented. For GeoMx this refers to whether segmentation was used to split ROIs (regions of interest) into AOIs (areas of interest). | ['Yes', 'No'] | True | +| roi_segmentation_strategy | Textfield | The method of segmentation that was applied in a GeoMx assay. If an overlay was used the overlay image needs to be included in the dataset upload. | | False | +| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | +| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | +| targeted_entity_label | Textfield | State what cell type(s) or functional tissue unit was targeted in this ROI/AOI. | | True | +| targeted_entity_id | Textfield | The ontology ID for the targeted entity. | | False | +| segment_id | Textfield | This is the ID for the area of interest (AOI) in a GeoMx dataset. From "Initial Dataset" spreadsheet (download from within Data Analysis Suite), e.g. 9a828e39-43d8-4051-9bcc-581a520a85d4. | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ['Yes', 'No'] | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | True | + +
+ +
nCounter Version 2 (current) + +## nCounter Version 2 (current) + +| attribute | type | description | value | required | +|-----------------------------------------------------|----------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------|------------| +| dataset_type | Textfield | The specific type of dataset being produced. | | True | +| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Textfield | The time duration unit of measurement | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Textfield | The time unit of measurement | | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ['Yes', 'No'] | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | +| mapped_area_unit | Textfield | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | +| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | True | +| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | True | +| target_retrieval_incubation_time_unit | Textfield | The units for target retrieval incubation time value. | | True | +| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | False | +| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | False | +| proteinasek_incubation_time_unit | Textfield | The units for proteinaseK incubation time value. | | False | +| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | True | +| is_roi_segmentation_performed | Allowable Value | Was the image segmented. For GeoMx this refers to whether segmentation was used to split ROIs (regions of interest) into AOIs (areas of interest). | ['Yes', 'No'] | True | +| roi_segmentation_strategy | Textfield | The method of segmentation that was applied in a GeoMx assay. If an overlay was used the overlay image needs to be included in the dataset upload. | | False | +| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | +| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | +| targeted_entity_label | Textfield | State what cell type(s) or functional tissue unit was targeted in this ROI/AOI. | | True | +| targeted_entity_id | Textfield | The ontology ID for the targeted entity. | | False | +| segment_id | Textfield | This is the ID for the area of interest (AOI) in a GeoMx dataset. From "Initial Dataset" spreadsheet (download from within Data Analysis Suite), e.g. 9a828e39-43d8-4051-9bcc-581a520a85d4. | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ['Yes', 'No'] | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| hybcode_pack_lot_number | Textfield | Enter the lot number noted within the LabWorksheet.txt file (and used in downstream nCounter processing). | | True | +| probe_hybridization_time_value | Numeric | How many hours were the oligo-conjugated RNA or oligo-conjugated antibody probes hybridized with the sample? | | True | +| probe_hybridization_time_unit | Textfield | The units for probe hybridization time value. | | True | +| oligo_probe_panel | Textfield | This is the probe panel used to target genes and/or proteins. In cases where there is a core panel and add-on modules, the core panel should be selected here. If additional panels are used, then they must be included in the "additional_panels_used.csv" file that's uploaded with the dataset. | | True | +| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ['Yes', 'No'] | True | +| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | True | + +
diff --git a/docs/assays/metadata/HiFi.md b/docs/assays/metadata/HiFi.md new file mode 100644 index 00000000..ca89eb5c --- /dev/null +++ b/docs/assays/metadata/HiFi.md @@ -0,0 +1,47 @@ +--- +layout: page +--- +# HiFi + +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | +| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | True | +| spot_size_value | Numeric | FModified progressive staining, Not applicable, Progressive staining, Regressive stainingor assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | +| spot_size_unit | Allowable Value | The unit for spot size value. | ```um^2``` ```mm^2``` | True | +| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | +| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | +| spot_spacing_unit | Allowable Value | Units corresponding to inter-spot distance | ```um``` | True | +| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be ```A1, B1, C1, D1```. For HiFi this would be the lane on the flowcell. | ```A1``` ```B1``` ```C1``` ```D1``` ```Lane 1``` ```Lane 2``` ```Lane 3``` ```Lane 4``` ```Lane 5``` ```Lane 6``` ```Lane 7``` ```Lane 8``` | False | +| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | True | +| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | True | +| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | True | +| target_retrieval_incubation_time_unit | Allowable Value | The units for target retrieval incubation time value. | ```minute``` | True | +| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | True | +| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | True | +| proteinasek_incubation_time_unit | Allowable Value | The units for proteinaseK incubation time value. | ```minute``` | True | +| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | +| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | False | + +
diff --git a/docs/assays/metadata/Histology.md b/docs/assays/metadata/Histology.md index b02db691..7170d0bb 100644 --- a/docs/assays/metadata/Histology.md +++ b/docs/assays/metadata/Histology.md @@ -1,68 +1,41 @@ ---- -layout: page-triary ---- - -# Histology Metadata Attributes - -Fields that are collected for Histology data, available at ```dataset.metadata.``` -  - -* indicates a required field - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | -| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | -| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | -| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | -| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | -| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | -| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | -| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | -| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | -| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | -| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | -| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | -| is_image_preprocessing_required | | Indicates whether image preprocessing is necessary based on the type of acquisition instrument used, such as a microscope or slide scanner. This may involve steps like fusing image tiles to assemble the complete image. Example: Yes | | -| stain_name *| | The name of the chemical stains (dyes) applied to histology samples to highlight important features of the tissue as well as to enhance the tissue contrast. | ```AB-PAS``` ```H&E``` ```H-DAB``` ```LFB``` ```PAS``` ```Trichrome``` | -| stain_technique | | There are typically three types of stains: progressive, modified progressive, and regressive. Progressive staining occurs when the hematoxylin is added to the tissue without being followed by a differentiator to remove excess dye. With regressive and modified progressive staining, a differentiator is used. | ```Modified progressive staining``` ```Not applicable``` ```Progressive staining``` ```Regressive staining``` | -| is_batch_staining_done *| | Are the slides stained using a linear batch method or individually? | | -| is_staining_automated *| | Is the slide staining automated with an instrument? | | -| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | -| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | -| slide_id | | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | -| tile_configuration | | The configuration of tiles used for stitching in the assay process. If no tile configuration is applicable, enter "Not applicable". Example: Row-by-row | ```Column-by-column``` ```Not applicable``` ```Snake-by-columns``` ```Row-by-row``` ```Snake-by-rows``` | -| scan_direction | | The direction of imaging, which is necessary for the stitching process. Example: Left-and-down | ```Left-and-down``` ```Right-and-down``` ```Not applicable``` ```Right-and-up``` ```Left-and-up``` | -| tiled_image_columns | | The number of columns used in the stitching process of a tiled image, often referred to as the grid size in the x-dimension. Example: 5 | | -| tiled_image_count | | The total number of raw tiled images captured, which are intended to be stitched together. Example: 75 | | -| intended_tile_overlap_percentage | | The intended percentage of overlap between tiled images. This value serves as the set point, although slight variations may occur during image acquisition due to stage registration. Example: 5 | | -| non_global_files | | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | -| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | - - -  - -## Deprecated Attributes - - These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. - - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | -| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | -| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | -| operator | | Name of the person responsible for executing the assay. | | -| operator_email | | Email address for the operator. | | -| overall_protocols_io_doi | | DOI for protocols.io referring to the overall protocol for the assay. | | -| principal_investigator | | | | -| pi_email | | Email address for the principal investigator. | | -| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | -| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | -| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | -| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | -| resolution_z_unit | | The unit of incremental distance between image slices. | ```mm``` ```um``` ```nm``` | -| resolution_z_value | | Optional if assay does not have multiple z-levels. Note that thisis resolution within a given sample: z-pitch (resolution_z_value) is the incrementdistance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stageis moved up or down in increments of 1.5um to capture images of several focalplanes. The best one will be used & the rest discarded. The thickness of the sampleitself is sample metadata. | | -| version | | Version of the schema to use when validating this metadata. | ```1``` | +--- +layout: page +--- +# Histology + +
Version 2 (Latest) + +## Version 2 (Latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | +| is_batch_staining_done | Allowable Value | Are the slides stained using a linear batch method or individually? | ```Yes``` ```No``` | True | +| is_staining_automated | Allowable Value | Is the slide staining automated with an instrument? | ```Yes``` ```No``` | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | +| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | +| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| stain_name | Allowable Value | The name of the chemical stains (dyes) applied to histology samples to highlight important features of the tissue as well as to enhance the tissue contrast. | ```AB-PAS``` ```H&E``` ```H-DAB``` ```LFB``` ```PAS``` ```Trichrome ```| True | +| stain_technique | Allowable Value | There are typically three types of stains: progressive, modified progressive, and regressive. Progressive staining occurs when the hematoxylin is added to the tissue without being followed by a differentiator to remove excess dye. With regressive and modified progressive staining, a differentiator is used. | ```Modified progressive staining``` ```Not applicable``` ```Progressive staining``` ```Regressive staining``` | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | +| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | + +
diff --git a/docs/assays/metadata/IMC.md b/docs/assays/metadata/IMC.md new file mode 100644 index 00000000..d4383200 --- /dev/null +++ b/docs/assays/metadata/IMC.md @@ -0,0 +1,234 @@ +--- +layout: page +--- +# IMC-2D + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
2D IMC Version 2 (Latest) + +## 2D IMC Version 2 (Latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| total_run_time_value | Numeric | How long the tissue was on the acquisition instrument. | | True | +| total_run_time_unit | Allowable Value | The units for the total run time unit field. | ```Hour``` ```Minute``` | True | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| data_precision_bytes | Numeric | Numerical data precision in bytes. | | True | +| ablation_frequency_value | Numeric | Frequency value of laser ablation | | True | +| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ```Hz``` | True | +| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | + + +
+ +
2D IMC Version 1 + +## 2D IMC Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['Imaging Mass Cytometry'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | +| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | +| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | +| number_of_channels | Numeric | Number of mass channels measured | | True | +| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | +| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | +| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | +| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | +| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | +| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | +| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | +| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | +| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | +| dual_count_start | Numeric | Threshold for dual counting. | | True | +| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | +| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | +| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | +| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | +| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | +| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | +| data_precision_bytes | Numeric | Numerical data precision in bytes | | True | +| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
2D IMC Version 0 + +## 2D IMC Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['Imaging Mass Cytometry'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | +| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | +| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | +| number_of_channels | Numeric | Number of mass channels measured | | True | +| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | +| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | +| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | +| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | +| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | +| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | +| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | +| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | +| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | +| dual_count_start | Numeric | Threshold for dual counting. | | True | +| end_datetime | Datetime | Time stamp indicating end of ablation for ROI | | True | +| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | +| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | +| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | +| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | +| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | +| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | +| start_datetime | Datetime | Time stamp indicating start of ablation for ROI | | True | +| data_precision_bytes | Numeric | Numerical data precision in bytes | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
3D IMC Version 1 (no longer accepting data) + +## 3D IMC Version 1 (no longer accepting data) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['3D Imaging Mass Cytometry'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | +| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | +| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | +| number_of_channels | Numeric | Number of mass channels measured | | True | +| number_of_sections | Numeric | Number of sections | | True | +| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | +| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | +| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | +| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | +| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | +| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | +| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | +| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | +| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | +| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | +| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | +| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | +| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | +| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | +| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | +| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
3D IMC Version 0 + +## 3D IMC Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['3D Imaging Mass Cytometry'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | +| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | +| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | +| number_of_channels | Numeric | Number of mass channels measured | | True | +| number_of_sections | Numeric | Number of sections | | True | +| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | +| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | +| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | +| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | +| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | +| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | +| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | +| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | +| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | +| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | +| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | +| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | +| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | +| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | +| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | +| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
diff --git a/docs/assays/metadata/Illumina-Spatial.md b/docs/assays/metadata/Illumina-Spatial.md new file mode 100644 index 00000000..498c6115 --- /dev/null +++ b/docs/assays/metadata/Illumina-Spatial.md @@ -0,0 +1,41 @@ +--- +layout: page +--- +# Illumina Spatial ver0 + +
Version 0 (current) + +## Version 0 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial v0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics``` | True | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | True | +| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | True | +| capture_area_id | Radio | The capture area on the slide that was used during the process. For example, in the case for Visium, this would correspond to areas such as [A1, B1, C1, D1], while for HiFi, it would refer to the lane on the flowcell. Example: A1 | ```A1```, ```B1```, ```C1```, ```D1```, ```Lane 1```, ```Lane 2```, ```Lane 3```, ```Lane 4```, ```Lane 5```, ```Lane 6```, ```Lane 7```, ```Lane 8``` | False | +| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | +| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | False | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| preparation_instrument_vendor | Assigned Value | The company that manufactures the instrument used to prepare the sample (e.g., for staining or other processing steps) prior to the assay. If the instrument was custom-built or developed internally, enter "In-House". If no sample preparation occurred, enter "Not applicable". Example: 10X Genomics | ```Thermo Fisher Scientific```, ```SunChrom```, ```Akoya Biosciences```, ```Leica Biosystems```, ```Ionpath```, ```Roche Diagnostics```, ```In-House```, ```Not applicable```, ```Hamamatsu```, ```HTX Technologies```, ```10x Genomics``` | False | +| preparation_instrument_model | Assigned Value | The specific model of the instrument used for sample preparation, such as staining. Manufacturers may offer multiple models with varying features or sensitivities, which can influence how the sample is processed and how the resulting data is interpreted. If no sample preparation occurred, enter "Not applicable". Example: Chromium X | ```AutoStainer XL```, ```ST5020 Multistainer```, ```Visium CytAssist```, ```SunCollect Sprayer```, ```Chromium X```, ```Chromium iX```, ```EVOS M7000```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```Discovery Ultra```, ```Sublimator```, ```Not applicable```, ```TM-Sprayer```, ```M5 Sprayer```, ```M3+ Sprayer```, ```Chromium Controller```, ```Chromium Connect```, ```Custom``` | False | +| capture_area_width_value | Numeric | The width of RNA capture area. Example: 10 | | True | +| capture_area_width_unit | Assigned Value | The unit of measurement for the capture area width value. If the width value is not specified, this field may be left blank. Example: mm | ```mm``` | True | +| capture_area_height_value | Numeric | The height of RNA capture area. Example: 10 | | True | +| capture_area_height_unit | Assigned Value | The unit of measurement for the capture area height value. If the height value is not specified, this field may be left blank. Example: mm | ```mm``` | True | +| spatial_discreatization_method | Assigned Value | The segmentation method used to divide the capture are into smaller, defined regions for analysis. Example: Cell segmentation | ```Square binning```, ```Cell segmentation```, ```Hexagonal binning``` | True | +| bin_size | Textfield | The size (in µm) of each discrete spatial unit ("bin") used to partition the capture area in bin-based spatial discretization. Example: 100 | | False | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | + +
\ No newline at end of file diff --git a/docs/assays/metadata/LC-MS.md b/docs/assays/metadata/LC-MS.md index a98bf37b..75cdce06 100644 --- a/docs/assays/metadata/LC-MS.md +++ b/docs/assays/metadata/LC-MS.md @@ -1,89 +1,296 @@ ---- -layout: page-triary ---- - -# LC-MS Metadata Attributes - -Fields that are collected for LC-MS data, available at ```dataset.metadata.``` -  - -* indicates a required field - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | -| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | -| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | -| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | -| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | -| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | -| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | -| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | -| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | -| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | -| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | -| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | -| ms_ionization_technique *| | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20``` | -| ms_scan_mode *| | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 for TMT) | ```MS1``` ```MS2``` ```MS3``` | -| mass_analysis_polarity *| | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | -| mass_to_charge_range_low_value | | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | -| mass_to_charge_range_high_value | | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | -| mass_resolving_power | | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | -| mass_to_charge_resolving_power | | The peak (m/z) used to calculate the resolving power. | | -| ion_mobility | | Specifies whether or not ion mobility spectrometry was performed and which technology was used. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ```TIMS``` ```SLIM``` ```FAIMS``` ```DTIMS``` ```cIMS``` ```TWIMS``` | -| data_collection_mode *| | Mode of data collection in tandem MS assays. Either DDA (Data-dependent acquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring), or PRM (parallel reaction monitoring). | ```DDA``` ```PRM``` ```DIA``` ```SRM``` | -| label_name | | If the samples were labeled (e.g. TMT), provide the name/ID of the label on this sample. | | -| lc_instrument_vendor | | The manufacturer of the instrument used for LC | ```Thermo Fisher Scientific``` ```Sciex``` ```In-House``` ```Agilent Technologies``` ```Waters``` ```Bruker``` ```Evosep``` | -| lc_instrument_model | | The model number/name of the instrument used for LC | | -| lc_column_vendor | | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulled tip capilary is used | ```Thermo Fisher Scientific``` ```In-House``` ```Waters``` ```Bruker``` ```Evosep``` ```IonOpticks``` | -| lc_column_model | | The model number/name of the LC Column - IF custom self-packed, pulled tip calillary is used enter "Pulled tip capilary" | | -| lc_resin | | Details of the resin used for lc, including vendor, particle size, pore size | | -| lc_column_length_value | | Liquid chromatography column length. | | -| lc_column_length_unit | | Units for liquid chromatography column length (typically cm). | ```um``` ```mm``` ```cm``` | -| lc_temperature_value | | Liquid chromatography temperature. | | -| lc_temperature_unit | | | ```celsius``` | -| lc_inner_diameter_value | | Liquid chromatography column inner diameter. | | -| lc_inner_diameter_unit | | | ```um``` ```mm``` ```cm``` | -| lc_flow_rate_value | | Value of flow rate. | | -| lc_flow_rate_unit | | Units of flow rate. | ```nL/min``` ```mL/min``` | -| lc_gradient_value | | Liquid chromatography gradient. | | -| lc_gradient_unit | | Unit for liquid chromatography gradient | ```minute``` | -| lc_mobile_phase_a | | Composition of mobile phase A | | -| lc_mobile_phase_b | | Composition of mobile phase B | | -| spatial_sampling_technique | | | ```nanoSPLITS``` ```nanoPOTS``` ```LESA``` ```microPOTS``` ```LCM``` ```microLESA``` | -| spatial_sampling_target | | Specifies the cell-type or functional tissue unit (FTU) that is targeted in the spatial profiling experiment. Leave blank if data are generated in imaging mode without a specific target structure. | | -| spatial_sampling_type | | Specifies whether or not the analysis was performed in a spatially targeted manner. Spatial profiling experiments target specific tissue foci but do not necessarily generate images. Spatial imaging expriments collect data from a regular array (pixels) that can be visualized as heat maps of ion intensity at each location (molecular images). Leave blank if data are derived from bulk analysis. | ```Imaging``` ```Profiling``` | -| analysis_protocol_doi *| | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | -| acquisition_protocol_doi | | | | -| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | - - -  - -## Deprecated Attributes - - These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. - - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | -| description | | Free-text description of this assay. | | -| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | -| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | -| operator | | Name of the person responsible for executing the assay. | | -| operator_email | | Email address for the operator. | | -| protocols_io_doi | | DOI for protocols.io referring to the protocol for this assay. | | -| overall_protocols_io_doi | | DOI for protocols.io for the overall process for this assay. | | -| pi | | Name of the principal investigator responsible for the data. | | -| pi_email | | Email address for the principal investigator. | | -| processing_search | | Software for analyzing and searching LC-MS/MS omics data | | -| labeling | | Indicates whether samples were labeled prior to MS analysis (e.g., TMT) | | -| dms | | Was differential mobility spectrometry used in this assay? | | -| resolution_x_unit | | The unit of measurement of the width of a pixel. | ```mm``` ```um``` ```nm``` | -| resolution_x_value | | The width of a pixel. | | -| resolution_y_unit | | The unit of measurement of the height of a pixel. | ```mm``` ```um``` ```nm``` | -| resolution_y_value | | The height of a pixel | | -| version | | Version of the schema to use when validating this metadata. | ```1``` | +--- +layout: page +--- +# LC-MS + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
Version 4 (Latest) + +## Version 4 (Latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | +| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | +| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | +| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | False | +| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | +| ion_mobility | Allowable Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```cIMS``` ```DTIMS``` ```FAIMS``` ```SLIM``` ```TIMS``` ```TWIMS``` | False | +| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | +| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | +| label_name | Textfield | If the samples were labeled (e.g. TMT), provide the name/ID of the label on this sample. Leave blank if not applicable. | | False | +| lc_instrument_vendor | Allowable Value | The manufacturer of the instrument used for liquid chromatography. | ```Agilent Technologies``` ```Bruker``` ```Evosep``` ```In-House``` ```Sciex``` ```Thermo Fisher Scientific``` ```Waters``` | False | +| lc_instrument_model | Textfield | The model number/name of the instrument used for liquid chromatography. | | False | +| lc_column_model | Textfield | The model number/name of the liquid chromatography column. If it is a custom self-packed, pulled tip capillary is used enter “Pulled tip capilary”. | | False | +| lc_resin | Textfield | Details of the resin used for liquid chromatography, including vendor, particle size, pore size. | | False | +| lc_column_length_value | Numeric | Liquid chromatography column length. | | False | +| lc_column_length_unit | Allowable Value | Units for liquid chromatography column length (typically cm). | ```um``` ```mm``` ```cm``` | False | +| lc_temperature_value | Numeric | Liquid chromatography temperature. | | False | +| lc_inner_diameter_value | Numeric | Liquid chromatography column inner diameter. | | False | +| lc_flow_rate_value | Numeric | Value of flow rate. | | False | +| lc_gradient_value | Numeric | Liquid chromatography gradient. | | False | +| lc_gradient_unit | Allowable Value | Unit for liquid chromatography gradient | ```Minute``` | False | +| lc_mobile_phase_a | Textfield | Composition of mobile phase A. | | False | +| lc_mobile_phase_b | Textfield | | | False | +| spatial_sampling_technique | Allowable Value | | ```LCM``` ```LESA``` ```microLESA``` ```microPOTS``` ```nanoPOTS``` ```nanoSPLITS``` | False | +| spatial_sampling_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targeted in the spatial profiling experiment. Leave blank if data are generated in imaging mode without a specific target structure. | | False | +| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependent acquisition), DIA (Data-independent acquisition), SRM (multiple reaction monitoring), or PRM (parallel reaction monitoring). | ```DDA``` ```PRM``` ```DIA``` ```SRM``` | False | +| lc_column_vendor | Allowable Value | The manufacturer of the liquid chromatography column unless self-packed, pulled tip capillary is used. | ```Bruker``` ```Evosep``` ```In-House``` ```IonOpticks``` ```Thermo Fisher Scientific``` ```Waters``` | False | +| lc_temperature_unit | Allowable Value | | ```Celsius``` | False | +| lc_inner_diameter_unit | Allowable Value | | ```um``` ```mm``` ```cm``` | False | +| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ```mL/min``` ```nL/min``` | False | +| spatial_sampling_type | Allowable Value | Specifies whether or not the analysis was performed in a spatially targeted manner. Spatial profiling experiments target specific tissue foci but do not necessarily generate images. Spatial imaging expriments collect data from a regular array (pixels) that can be visualized as heat maps of ion intensity at each location (molecular images). Leave blank if data are derived from bulk analysis. | ```Imaging``` ```Profiling``` | False | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
+ +
Version 3 + +## Version 3 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['3'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | +| assay_type | Allowable Value | Bottom-up refers to analyzing proteins in a sample by digesting themto peptides. Top-down refers to analyzing whole proteins without digestion. LC-MSand MS are for lipids/metabolites. LC-MS Bottom-Up and MS Bottom-Up are for peptides.LC-MS Top-Down and MS Top-Down are for proteins. | ['LC-MS', 'MS', 'LC-MS Bottom-Up', 'MS Bottom-Up', 'LC-MS Top-Down', 'MS Top-Down'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids', 'peptides', 'phosphopeptides', 'glycans'] | False | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | +| dms | Allowable Value | Was differential mobility spectrometry used in this assay? | ['Yes','No'] | True | +| ms_source | Allowable Value | The ion source type used for surface sampling. | ['ESI'] | True | +| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | +| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | +| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | +| mass_resolving_power | Numeric | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | False | +| mz_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | +| ion_mobility | Allowable Value | Specifies whether or not ion mobility spectrometry was performed andwhich technology was used. Technologies for measuring ion mobility: TravelingWave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS),High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube IonMobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ['TIMS', 'TWIMS', 'FAIMS', 'DTIMS', 'SLIMS'] | False | +| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | +| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | +| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | +| label_name | Textfield | If the samples were labeled (e.g. TMT), provide the name/ID of thelabel on this sample. | | False | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | +| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | +| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | +| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | +| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | +| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | +| lc_length_value | Numeric | LC column length | | False | +| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | +| lc_temp_value | Numeric | LC temperature | | False | +| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | +| lc_id_value | Numeric | LC column inner diameter (microns) | | False | +| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | +| lc_flow_rate_value | Numeric | Value of flow rate. | | False | +| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | +| lc_gradient | Textfield | LC gradient | | False | +| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | +| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | +| spatial_type | Allowable Value | Specifies whether or not the analysis was performed in a spatialy targetedmanner and the technique used for spatial sampling. For example, Laser-capturemicrodissection (LCM), Liquid Extraction Surface Analysis (LESA), NanodropletProcessing in One pot for Trace Samples (nanoPOTS). | ['LCM', 'LESA', 'nanoPOTS', 'microLESA'] | False | +| spatial_sampling_type | Allowable Value | Specifies whether or not the analysis was performed in a spatiallytargeted manner. Spatial profiling experiments target specific tissue foci butdo not necessarily generate images. Spatial imaging expriments collect data froma regular array (pixels) that can be visualized as heat maps of ion intensityat each location (molecular images). Leave blank if data are derived from bulkanalysis. | ['profiling', 'imaging'] | False | +| spatial_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targetedin the spatial profiling experiment. Leave blank if data are generated in imagingmode without a specific target structure. | | False | +| resolution_x_value | Numeric | The width of a pixel. | | False | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | False | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | +| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | +| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | + +
+ +
Version 2 + +## Version 2 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | +| assay_type | Allowable Value | Bottom-up refers to analyzing proteins in a sample by digesting themto peptides. Top-down refers to analyzing whole proteins without digestion. LC-MSand MS are for lipids/metabolites. LC-MS Bottom-Up and MS Bottom-Up are for peptides.LC-MS Top-Down and MS Top-Down are for proteins. | ['LC-MS', 'MS', 'LC-MS Bottom-Up', 'MS Bottom-Up', 'LC-MS Top-Down', 'MS Top-Down'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids', 'peptides', 'phosphopeptides', 'glycans'] | False | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | +| ms_source | Allowable Value | The ion source type used for surface sampling. | ['ESI'] | True | +| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | +| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | +| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | +| mass_resolving_power | Numeric | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | False | +| mz_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | +| ion_mobility | Allowable Value | Specifies whether or not ion mobility spectrometry was performed andwhich technology was used. Technologies for measuring ion mobility: TravelingWave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS),High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube IonMobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ['TIMS', 'TWIMS', 'FAIMS', 'DTIMS', 'SLIMS'] | False | +| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | +| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | +| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | +| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | +| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | +| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | +| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | +| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | +| lc_length_value | Numeric | LC column length | | False | +| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | +| lc_temp_value | Numeric | LC temperature | | False | +| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | +| lc_id_value | Numeric | LC column inner diameter (microns) | | False | +| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | +| lc_flow_rate_value | Numeric | Value of flow rate. | | False | +| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | +| lc_gradient | Textfield | LC gradient | | False | +| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | +| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | +| spatial_type | Allowable Value | Specifies whether or not the analysis was performed in a spatialy targetedmanner and the technique used for spatial sampling. For example, Laser-capturemicrodissection (LCM), Liquid Extraction Surface Analysis (LESA), NanodropletProcessing in One pot for Trace Samples (nanoPOTS). | ['LCM', 'LESA', 'nanoPOTS', 'microLESA'] | False | +| spatial_sampling_type | Allowable Value | Specifies whether or not the analysis was performed in a spatiallytargeted manner. Spatial profiling experiments target specific tissue foci butdo not necessarily generate images. Spatial imaging expriments collect data froma regular array (pixels) that can be visualized as heat maps of ion intensityat each location (molecular images). Leave blank if data are derived from bulkanalysis. | ['profiling', 'imaging'] | False | +| spatial_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targetedin the spatial profiling experiment. Leave blank if data are generated in imagingmode without a specific target structure. | | False | +| resolution_x_value | Numeric | The width of a pixel. | | False | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | False | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | +| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | +| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | + +
+ +
Version 1 + +## Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['LC-MS (metabolomics)', 'LC-MS/MS (label-free proteomics)', 'MS (shotgun lipidomics)'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | False | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | +| ms_source | Textfield | The ion source type used for surface sampling (MALDI, MALDI-2, DESI,or SIMS) or LC-MS/MS data acquisition (nESI) | | True | +| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | +| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | +| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | +| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | +| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | +| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | +| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | +| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | +| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | +| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | +| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | +| lc_length_value | Numeric | LC column length | | False | +| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | +| lc_temp_value | Numeric | LC temperature | | False | +| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | +| lc_id_value | Numeric | LC column inner diameter (microns) | | False | +| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | +| lc_flow_rate_value | Numeric | Value of flow rate. | | False | +| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | +| lc_gradient | Textfield | LC gradient | | False | +| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | +| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | +| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | +| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | +| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | + +
+ +
Version 0 + +## Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['LC-MS (metabolomics)', 'LC-MS/MS (label-free proteomics)', 'MS (shotgun lipidomics)'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | False | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | +| ms_source | Textfield | The ion source type used for surface sampling (MALDI, MALDI-2, DESI,or SIMS) or LC-MS/MS data acquisition (nESI) | | True | +| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | +| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | +| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | +| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | +| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | +| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | +| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | +| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | +| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | +| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | +| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | +| lc_length_value | Numeric | LC column length | | False | +| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | +| lc_temp_value | Numeric | LC temperature | | False | +| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | +| lc_id_value | Numeric | LC column inner diameter (microns) | | False | +| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | +| lc_flow_rate_value | Numeric | Value of flow rate. | | False | +| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | +| lc_gradient | Textfield | LC gradient | | False | +| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | +| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | +| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | +| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | +| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/LightSheet.md b/docs/assays/metadata/LightSheet.md new file mode 100644 index 00000000..4eb8950d --- /dev/null +++ b/docs/assays/metadata/LightSheet.md @@ -0,0 +1,146 @@ +--- +layout: page +--- +# Light-Sheet + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
Version 3 (latest) + +## Version 3 (latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | +| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | +| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
+ +
Version 2 + +## Version 2 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['Light Sheet'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | +| resolution_x_value | Numeric | The width of a pixel. | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| range_z_value | Numeric | The total range of the z axis. | | True | +| range_z_unit | Allowable Value | The unit of range_z_value. | ['nm', 'um'] | False | +| step_z_value | Numeric | The number of optical sections in z axis range. | | True | +| increment_z_value | Numeric | The distance between sequential optical sections. | | True | +| increment_z_unit | Allowable Value | The units of increment z value. | ['nm', 'um'] | False | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | +| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | + +
+ +
Version 1 + +## Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['Light Sheet'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | +| resolution_x_value | Numeric | The width of a pixel. | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| resolution_z_value | Numeric | The distance at which two objects along the detection z-axis can bedistinguished (resolved as 2 objects). | | True | +| resolution_z_unit | Allowable Value | The unit of distance at which two objects along the detection z-axiscan be distinguished (resolved as 2 objects). | ['mm', 'um', 'nm'] | False | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | +| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | + +
+ +
Version 0 + +## Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['Light Sheet'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | +| resolution_x_value | Numeric | The width of a pixel. | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| resolution_z_value | Numeric | The distance at which two objects along the detection z-axis can bedistinguished (resolved as 2 objects). | | True | +| resolution_z_unit | Allowable Value | The unit of distance at which two objects along the detection z-axiscan be distinguished (resolved as 2 objects). | ['mm', 'um', 'nm'] | False | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | +| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | + +
diff --git a/docs/assays/metadata/MALDI.md b/docs/assays/metadata/MALDI.md index ae760058..ccf79e7c 100644 --- a/docs/assays/metadata/MALDI.md +++ b/docs/assays/metadata/MALDI.md @@ -1,67 +1,171 @@ ---- -layout: page-triary ---- - -# MALDI Metadata Attributes - -Fields that are collected for MALDI data, available at ```dataset.metadata.``` -  - -* indicates a required field - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | -| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | -| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | -| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | -| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | -| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | -| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | -| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | -| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | -| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | -| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | -| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | -| ms_ionization_technique *| | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20``` | -| ms_scan_mode *| | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | -| mass_analysis_polarity *| | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | -| mass_to_charge_range_low_value | | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | -| mass_to_charge_range_high_value | | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | -| mass_resolving_power *| | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | -| mass_to_charge_resolving_power | | The peak (m/z) used to calculate the resolving power. | | -| ion_mobility | | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```TIMS``` ```SLIM``` ```FAIMS``` ```DTIMS``` ```cIMS``` ```TWIMS``` | -| matrix_deposition_method *| | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | -| preparation_instrument_vendor *| | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | -| preparation_instrument_model *| | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | -| preparation_matrix *| | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | -| analysis_protocol_doi *| | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | -| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | - - -  - -## Deprecated Attributes - - These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. - - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | -| description | | Free-text description of this assay. | | -| section_prep_protocols_io_doi | | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | -| overall_protocols_io_doi | | DOI for protocols.io referring to the overall protocol for the assay. | | -| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | -| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | -| operator | | Name of the person responsible for executing the assay. | | -| operator_email | | Email address for the operator. | | -| pi | | Name of the principal investigator responsible for the data. | | -| pi_email | | Email address for the principal investigator. | | -| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | -| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | -| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | -| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | -| version | | Version of the schema to use when validating this metadata. | ```1``` | +--- +layout: page +--- +# MALDI + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
Maldi Version 2 (latest) + +## Maldi Version 2 (latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | +| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | True | +| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | +| ion_mobility | Allowable Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```cIMS``` ```DTIMS``` ```FAIMS``` ```SLIM``` ```TIMS``` ```TWIMS``` | False | +| matrix_deposition_method | Allowable Value | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | True | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | +| preparation_matrix | Allowable Value | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | +| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | +| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | +| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | +| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
+ +
IMS Version 2 + +## IMS Version 2 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['MALDI-IMS', 'SIMS-IMS', 'NanoDESI', 'DESI'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids', 'peptides', 'phosphopeptides', 'glycans'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| ms_source | Allowable Value | The ion source type used for surface sampling (MALDI, MALDI-2, DESI, nanoDESI or SIMS). | ['MALDI', 'MALDI-2', 'LDI', 'LA', 'SIMS-C60', 'SIMS-H2O', 'DESI', 'nanoDESI'] | True | +| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | +| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | +| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | +| mass_resolving_power | Numeric | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | True | +| mz_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | +| ion_mobility | Allowable Value | Specifies whether or not ion mobility spectrometry was performed and which technology was used. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ['TIMS', 'TWIMS', 'FAIMS', 'DTIMS', 'SLIMS'] | False | +| ms_scan_mode | Allowable Value | Scan mode refers to the number of steps in the separation of fragments. | ['MS', 'MS/MS', 'MS3'] | True | +| resolution_x_value | Numeric | The width of a pixel. | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| preparation_type | Textfield | Common methods of depositing matrix for MALDI imaging include robotic spotting, electrospray deposition, and spray-coating with an airbrush. | | False | +| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | False | +| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | False | +| preparation_maldi_matrix | Textfield | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the laser. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | | False | +| desi_solvent | Textfield | Solvent composition for conducting nanospray desorption electrospray ionization (nanoDESI) or desorption electrospray ionization (DESI). | | False | +| desi_solvent_flow_rate | Numeric | The rate of flow of the solvent into a spray. | | False | +| desi_solvent_flow_rate_unit | Allowable Value | Units of the rate of solvent flow. | ['uL/minute'] | False | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | +| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | +| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
IMS Version 1 + +## IMS Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['MALDI-IMS'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| ms_source | Allowable Value | The ion source type used for surface sampling (MALDI, MALDI-2, DESI, or SIMS) or LC-MS/MS data acquisition (nESI) | ['MALDI', 'MALDI-2', 'DESI', 'SIMS', 'nESI'] | True | +| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | +| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | +| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | +| resolution_x_value | Numeric | The width of a pixel. | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| preparation_type | Textfield | Common methods of depositing matrix for MALDI imaging include robotic spotting, electrospray deposition, and spray-coating with an airbrush. | | True | +| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | +| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | +| preparation_maldi_matrix | Textfield | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the laser. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | +| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
IMS Version 0 + +## IMS Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['MALDI-IMS'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| ms_source | Allowable Value | The ion source type used for surface sampling (MALDI, MALDI-2, DESI, or SIMS) or LC-MS/MS data acquisition (nESI) | ['MALDI', 'MALDI-2', 'DESI', 'SIMS', 'nESI'] | True | +| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | +| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | +| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | +| resolution_x_value | Numeric | The width of a pixel. | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | +| resolution_y_value | Numeric | The height of a pixel | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | +| preparation_type | Textfield | Common methods of depositing matrix for MALDI imaging include robotic spotting, electrospray deposition, and spray-coating with an airbrush. | | True | +| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | +| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | +| preparation_maldi_matrix | Textfield | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the laser. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | +| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
diff --git a/docs/assays/metadata/MERFISH.md b/docs/assays/metadata/MERFISH.md new file mode 100644 index 00000000..bb457b7d --- /dev/null +++ b/docs/assays/metadata/MERFISH.md @@ -0,0 +1,47 @@ +--- +layout: page +--- +# MERFISH +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | False | +| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | False | +| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | +| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | False | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | False | +| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | False | +| target_retrieval_incubation_time_unit | Allowable Value | The units for target retrieval incubation time value. | ```minute``` | False | +| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | False | +| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | False | +| proteinasek_incubation_time_unit | Allowable Value | The units for proteinaseK incubation time value. | ```minute``` | False | +| probe_hybridization_time_value | Numeric | How long was the oligo-conjugated RNA or oligo-conjugated antibody probes hybridized with the sample? | | False | +| probe_hybridization_time_unit | Allowable Value | The units for probe hybridization time value. | ```Hour``` ```Minute``` | False | +| oligo_probe_panel | Allowable Value | This is the probe panel used to target genes and/or proteins. In cases where there is a core panel and add-on modules, the core panel should be selected here. If additional panels are used, then they must be included in the "additional_panels_used.csv" file that's uploaded with the dataset. | ```10x Genomics; Chromium Fixed RNA Kit``` ```Human Transcriptome``` ```4 rxns x 1 BC; PN 1000474``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```16 rxns; PN 1000420``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```64 rxns; PN 1000456``` ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363``` ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365``` ```Custom``` ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-HuWTA-4``` ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-MsWTA-4``` | True | +| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ```Yes``` ```No``` | True | +| number_of_panel_targets | Numeric | How many genes, RNA isoforms or RNA regions are targeted by probes. | | True | +| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | False | +| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | +| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | +| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | +| number_of_additional_stains | Numeric | This would be minimally 2 (always include DAPI and polyT) and can include 6 more. | | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | + +
diff --git a/docs/assays/metadata/MIBI.md b/docs/assays/metadata/MIBI.md index 6e500c70..6c06a9b1 100644 --- a/docs/assays/metadata/MIBI.md +++ b/docs/assays/metadata/MIBI.md @@ -1,86 +1,104 @@ ---- -layout: page-triary ---- - -# MIBI Metadata Attributes - -Fields that are collected for MIBI data, available at ```dataset.metadata.``` -  - -* indicates a required field - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | -| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | -| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | -| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | -| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | -| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | -| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | -| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | -| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | -| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | -| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | -| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | -| number_of_antibodies *| | Number of antibodies | | -| number_of_channels *| | Number of fluorescent channels imaged during each cycle. | | -| slide_id *| | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | -| roi_description *| | A description of the region of interest (ROI) captured in the image. | | -| roi_id *| | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | -| acquisition_id *| | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | -| area_normalized_ion_dose_value *| | Number of primary ions delivered to the sample per unit area | | -| area_normalized_ion_dose_unit *| | Area normalized ion dose unit | ```nA*hr/mm2``` | -| data_precision_bytes *| | Numerical data precision in bytes | | -| pixel_dwell_time_value *| | Resident time of primary ion beam on each pixel. | | -| pixel_dwell_time_unit *| | Pixel dwell time unit. | ```ms``` | -| antibodies_path *| | Relative path to file with antibody information for this dataset. | | -| primary_ion *| | Primary ion. | ```Xe``` | -| primary_ion_current_unit | | Primary ion current unit, typically nA or pA | ```nA``` ```pA``` | -| primary_ion_current_value *| | Primary ion current value. | | -| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | - - -  - -## Deprecated Attributes - - These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. - - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | -| description | | Free-text description of this assay. | | -| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | -| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | -| protocol_io_doi | | | | -| reagent_prep_protocols_io_doi | | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | -| preparation_instrument_model | | The model number/name of the instrument used to prepare the sample for the assay | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | -| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare the sample for the assay. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | -| operator | | Name of the person responsible for executing the assay. | | -| operator_email | | Email address for the operator. | | -| pi | | Name of the principal investigator responsible for the data. | | -| pi_email | | Email address for the principal investigator. | | -| segment_data_format | | This refers to the data type, which is a "float" for the IMC counts. | ```float``` ```integer``` ```string``` | -| signal_type | | Type of signal measured per channel (usually dual counts) | ```dual count``` ```pulse count``` ```intensity value``` | -| dual_count_start | | Threshold for dual counting. | | -| start_datetime | | Time stamp indicating start of ablation for ROI | | -| end_datetime | | Time stamp indicating end of ablation for ROI | | -| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | -| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | -| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | -| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | -| max_x_width_unit | | Units of image width of the ROI acquisition | ```um``` | -| max_x_width_value | | Image width value of the ROI acquisition | | -| max_y_height_unit | | Units of image height of the ROI acquisition | ```um``` | -| max_y_height_value | | Image height value of the ROI acquisition | | -| pixel_size_x_unit | | Width unit of the pixel or voxel measurement. | ```nm``` | -| pixel_size_x_value | | Width value of the pixel or voxel measurement (distinct from the image resolution_x_value). | | -| pixel_size_y_unit | | Length unit of the pixel or voxel measurement. | ```nm``` | -| pixel_size_y_value | | Length value of the pixel or voxel measurement (distinct from the image resolution_y_value). | | -| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | -| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | -| version | | Version of the schema to use when validating this metadata. | ```1``` | +--- +layout: page +--- +# MIBI + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
Version 2 (latest) + +## Version 2 (latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| roi_description | Textfield | A description of the anatomical structure being captured in the region of interest (ROI). | | True | +| roi_id | Numeric | Multiple images are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The ROI ID is a number from 1 to N representing the ROI captured on a slide. | | True | +| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the "Acquisition ID" and the "ROI ID" indicate the slide-ROI represented in the image. | | True | +| area_normalized_ion_dose_value | Numeric | Number of primary ions delivered to the sample per unit area. | | True | +| area_normalized_ion_dose_unit | Allowable Value | Area normalized ion dose unit. | ```nA*hr/mm2``` | True | +| data_precision_bytes | Numeric | Numerical data precision in bytes. | | True | +| pixel_dwell_time_value | Numeric | Resident time of primary ion beam on each pixel to ionize it. | | True | +| pixel_dwell_time_unit | Allowable Value | Pixel dwell time unit. | ```ms``` | True | +| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | + +
+ + +
Version 1 + +## Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|--------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['MIBI'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | +| resolution_x_value | Numeric | The width of a pixel. (Akoya pixel is 377nm square) | | True | +| resolution_x_unit | Allowable Value | The unit of measurement of width of a pixel.(nm) | ['mm', 'um', 'nm'] | False | +| resolution_y_value | Numeric | The height of a pixel. (Akoya pixel is 377nm square) | | True | +| resolution_y_unit | Allowable Value | The unit of measurement of height of a pixel. (nm) | ['mm', 'um', 'nm'] | False | +| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | +| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | +| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | +| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | +| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | +| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | +| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | +| area_normalized_ion_dose_unit | Allowable Value | Area normalized ion dose unit | ['nA*hr/mm2'] | False | +| area_normalized_ion_dose_value | Numeric | Number of primary ions delivered to the sample per unit area | | True | +| data_precision_bytes | Numeric | Numerical data precision in bytes | | True | +| dual_count_start | Numeric | Threshold for dual counting. | | True | +| end_datetime | Datetime | Time stamp indicating end of ablation for ROI | | True | +| pixel_dwell_time_value | Numeric | Resident time of primary ion beam on each pixel. | | True | +| pixel_dwell_time_unit | Allowable Value | Pixel dwell time unit. | ['ms'] | False | +| pixel_size_x_value | Numeric | Width value of the pixel or voxel measurement (distinct from the image resolution_x_value). | | True | +| pixel_size_x_unit | Allowable Value | Width unit of the pixel or voxel measurement. | ['nm'] | False | +| pixel_size_y_value | Numeric | Length value of the pixel or voxel measurement (distinct from the image resolution_y_value). | | True | +| pixel_size_y_unit | Allowable Value | Length unit of the pixel or voxel measurement. | ['nm'] | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare the sample for the assay. | ['Custom', 'Ionpath'] | True | +| preparation_instrument_model | Allowable Value | The model number/name of the instrument used to prepare the sample for the assay | ['Custom', 'MIBIscope 1', 'MIBIscope 2'] | True | +| primary_ion | Allowable Value | Primary ion. | ['Xe'] | True | +| primary_ion_current_value | Numeric | Primary ion current value. | | True | +| primary_ion_current_unit | Allowable Value | Primary ion current unit, typically nA or pA | ['nA', 'pA'] | False | +| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | +| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | +| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | +| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | +| start_datetime | Datetime | Time stamp indicating start of ablation for ROI | | True | +| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
diff --git a/docs/assays/metadata/MPLEx.md b/docs/assays/metadata/MPLEx.md new file mode 100644 index 00000000..d7c3f0b7 --- /dev/null +++ b/docs/assays/metadata/MPLEx.md @@ -0,0 +1,59 @@ +--- +layout: page +--- +# MPLEx + +
Version 2.0 (use this one) + +## Version 2.0 (use this one) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | +| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | +| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes```, ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| mass_analysis_polarity | Assigned Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode```, ```Positive ion mode```, ```Negative ion mode``` | True | +| mass_to_charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | +| mass_to_charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | +| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | False | +| mass_to_charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | +| ion_mobility | Assigned Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```TIMS```, ```SLIM```, ```FAIMS```, ```DTIMS```, ```cIMS```, ```TWIMS``` | False | +| ms_ionization_technique | Assigned Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```MALDI```, ```SIMS-C60```, ```LDI```, ```HESI```, ```nanoDESI```, ```MALDI-2```, ```DESI```, ```LA```, ```SIMS-H20```, ```ESI``` | True | +| ms_scan_mode | Assigned Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS2```, ```MS1```, ```MS3``` | True | +| label_name | Textfield | If the samples were labeled (e.g. TMT), provide the name/ID of the label on this sample. Leave blank if not applicable. | | False | +| lc_instrument_vendor | Assigned Value | The manufacturer of the instrument used for liquid chromatography. | ```Thermo Fisher Scientific```, ```Sciex```, ```In-House```, ```Agilent Technologies```, ```Waters```, ```Bruker```, ```Evosep``` | False | +| lc_instrument_model | Textfield | The model number/name of the instrument used for liquid chromatography. | | False | +| lc_column_model | Textfield | The model number/name of the liquid chromatography column. If it is a custom self-packed, pulled tip capillary is used enter “Pulled tip capilary”. | | False | +| lc_resin | Textfield | Details of the resin used for liquid chromatography, including vendor, particle size, pore size. | | False | +| lc_column_length_value | Numeric | Liquid chromatography column length. | | False | +| lc_column_length_unit | Assigned Value | Units for liquid chromatography column length (typically cm). | ```um```, ```mm```, ```cm``` | False | +| lc_temperature_value | Numeric | Liquid chromatography temperature. | | False | +| lc_inner_diameter_value | Numeric | Liquid chromatography column inner diameter. | | False | +| lc_flow_rate_value | Numeric | Value of flow rate. | | False | +| lc_gradient_value | Numeric | Liquid chromatography gradient. | | False | +| lc_gradient_unit | Assigned Value | Unit for liquid chromatography gradient | ```minute``` | False | +| lc_mobile_phase_a | Textfield | Composition of mobile phase A. | | False | +| lc_mobile_phase_b | Textfield | | | False | +| spatial_sampling_technique | Assigned Value | | ```nanoSPLITS```, ```nanoPOTS```, ```LESA```, ```microPOTS```, ```LCM```, ```microLESA``` | False | +| spatial_sampling_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targeted in the spatial profiling experiment. Leave blank if data are generated in imaging mode without a specific target structure. | | False | +| analysis_protocol_doi | Link | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| data_collection_mode | Assigned Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependent acquisition), DIA (Data-independent acquisition), SRM (multiple reaction monitoring), or PRM (parallel reaction monitoring). | ```DDA```, ```PRM```, ```DIA```, ```SRM``` | False | +| lc_column_vendor | Assigned Value | The manufacturer of the liquid chromatography column unless self-packed, pulled tip capillary is used. | ```Thermo Fisher Scientific```, ```In-House```, ```Waters```, ```Bruker```, ```Evosep```, ```IonOpticks``` | False | +| lc_temperature_unit | Assigned Value | | ```celsius``` | False | +| lc_inner_diameter_unit | Assigned Value | | ```um```, ```mm```, ```cm``` | False | +| lc_flow_rate_unit | Assigned Value | Units of flow rate. | ```nL/min```, ```mL/min``` | False | +| spatial_sampling_type | Assigned Value | Specifies whether or not the analysis was performed in a spatially targeted manner. Spatial profiling experiments target specific tissue foci but do not necessarily generate images. Spatial imaging expriments collect data from a regular array (pixels) that can be visualized as heat maps of ion intensity at each location (molecular images). Leave blank if data are derived from bulk analysis. | ```Imaging```, ```Profiling``` | False | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/MUSIC.md b/docs/assays/metadata/MUSIC.md index 48a650ac..58c71288 100644 --- a/docs/assays/metadata/MUSIC.md +++ b/docs/assays/metadata/MUSIC.md @@ -1,70 +1,58 @@ ---- -layout: page-triary ---- - -# MUSIC Metadata Attributes - -Fields that are collected for MUSIC data, available at ```dataset.metadata.``` -  - -* indicates a required field - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | -| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | -| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | -| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | -| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | -| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | -| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | -| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | -| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | -| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | -| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | -| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | -| barcode_offset *| | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | -| barcode_read *| | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | -| barcode_size *| | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | -| umi_offset *| | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | -| umi_read *| | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | -| umi_size *| | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | -| assay_input_entity *| | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | -| number_of_input_cells_or_nuclei | | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | -| amount_of_input_analyte_value | | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | -| amount_of_input_analyte_unit | | Units of amount of entity input to assay value | ```ug``` ```ng``` | -| library_adapter_sequence *| | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | -| library_average_fragment_size *| | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | -| library_input_amount_value | | The amount of cDNA, after amplification, that was used for library construction. | | -| library_input_amount_unit | | unit of library input amount value | ```ng``` ```ul``` | -| library_output_amount_value | | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | -| library_output_amount_unit | | Units of library final yield. | ```ng``` ```ul``` | -| library_concentration_value *| | The concentration value of the pooled library samples submitted for sequencing. | | -| library_concentration_unit *| | Unit of library concentration value. | ```ng/ul``` ```nM``` | -| library_layout *| | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | -| library_preparation_kit *| | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```1 slides``` ```4 reactions; PN 1000338``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 reactions; PN 1000187``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | -| sample_indexing_kit *| | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001``` | -| sample_indexing_set | | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | -| is_technical_replicate *| | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | | -| expected_entity_capture_count | | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | -| sequencing_reagent_kit *| | Reagent kit used for sequencing | ```Custom``` ```Illumina``` ```HiSeq 3000/4000 PE Cluster Kit PE-410-1001``` ```PN 1000283, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles)``` ```PN 20046811, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles)``` ```PN 20046812, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles)``` ```PN 20046813, Illumina``` ```NextSeq 2000 P3 Reagent Kit (300 Cycles)``` ```PN 20040561, Illumina``` ```NextSeq 2000 P3 Reagents Kit (100 Cycles)``` ```PN 20040559, Illumina``` ```NextSeq 500/550 Hi Output Kit 150 Cycles``` ```v2.5``` ```PN 20024907, Illumina``` ```NextSeq 500/550 Hi Output Kit 75 Cycles v2.5``` ```PN 20024906, Illumina``` ```NextSeq 500/550 Mid Output Kit 150 Cycles v2.5``` ```PN 20024904, Illumina``` ```NovaSeq 6000 S1 Reagent Kit (200 Cycles)``` ```PN 20012864, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028319, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028318, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles)``` ```PN 20028317, Illumina``` ```NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028316, Illumina``` ```NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles)``` ```PN 20028312, Illumina``` ```NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028313, Illumina``` ```NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles)``` ```PN 20028401, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (100 Cycle)``` ```PN 20104703, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (200 Cycle)``` ```PN 20104704, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (300 Cycle)``` ```PN 20104705, Illumina``` ```NovaSeq X Series 10B Reagent Kit (100 Cycle)``` ```PN 20085596, Illumina``` ```NovaSeq X Series 10B Reagent Kit (200 Cycle)``` ```PN 20085595, Illumina``` ```NovaSeq X Series 10B Reagent Kit (300 Cycle)``` ```PN 20085594``` | -| sequencing_read_format *| | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | -| sequencing_batch_id | | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | -| capture_batch_id | | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | -| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | -| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | -| preparation_instrument_kit | | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | -| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | - - -  - -## Deprecated Attributes - - These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. - - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| +--- +layout: page +--- +# MUSIC + +
Current Metadata Attributes + +## Current Metadata Attributes + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```14-17,14,14``` ```Not applicable``` | True | +| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | +| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | +| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | +| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | +| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | +| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | False | +| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | False | +| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | +| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | +| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | +| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | +| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | +| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | +| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | +| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | False | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | +| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | +| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | +| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | +| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | +| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```Not applicable``` | True | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,78``` ```10,48,86``` ```0,20-23,41-44``` ```Not applicable``` | True | +| amount_of_input_analyte_unit | Allowable Value | Units of amount of entity input to assay value | ```ug``` ```ng``` | False | + +
\ No newline at end of file diff --git a/docs/assays/metadata/Olink.md b/docs/assays/metadata/Olink.md new file mode 100644 index 00000000..2a36ed5b --- /dev/null +++ b/docs/assays/metadata/Olink.md @@ -0,0 +1,28 @@ +--- +layout: page +--- +# Olink + +
Version 2 (current) + +## Version 2 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | +| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | +| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | False | + +
\ No newline at end of file diff --git a/docs/assays/metadata/PhenoCycler.md b/docs/assays/metadata/PhenoCycler.md new file mode 100644 index 00000000..7883ef67 --- /dev/null +++ b/docs/assays/metadata/PhenoCycler.md @@ -0,0 +1,36 @@ +--- +layout: page +--- +# PhenoCycler + +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| number_of_antibodies | Numeric | Number of antibodies | | True | +| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | +| number_of_biomarker_imaging_rounds | Numeric | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo application, 2. fluor application, 3. washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | True | +| number_of_total_imaging_rounds | Numeric | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| total_run_time_value | Numeric | How long the tissue was on the acquisition instrument. | | True | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| total_run_time_unit | Allowable Value | The units for the total run time unit field. | ```Hour``` ```Minute``` | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| nuclear_marker_or_stain | Allowable Value | For markers, an antibody-targetted molecule present in or around the cell nucleus, the protein or gene symbol that identifies the antibody target that is used as the nuclear marker. This symbol must match the antibody target that is either generated from the panel used or entered with custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets this is the stain name (e.g., DAPI) and, when appropriate, associated staining kit and vendor. For the PhenoCycler, this symbol must match the value found in the XPD output file. | ```DAPI``` ```Not applicable``` | True | +| cell_boundary_marker_or_stain | Allowable Value | If a marker or stain was used to identify all cell boundaries in the tissue, then the name of the marker or stain should be included here. The name of the antibody-targeted molecule marker or non-antibody targeted molecule stain included here must be identical to what is found in the imaging data. For example, with the PhenoCycler, this name must match the value found in the XPD output file. If multiple marker or stains are used to identify all cell boundaries, then a comma separated list should be used here. | ```NAKATPASE``` ```CD298``` ```Not applicable``` | True | +| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | diff --git a/docs/assays/metadata/Pixel-seqV2.md b/docs/assays/metadata/Pixel-seqV2.md new file mode 100644 index 00000000..44e5a345 --- /dev/null +++ b/docs/assays/metadata/Pixel-seqV2.md @@ -0,0 +1,37 @@ +--- +layout: page +--- +# Pixel-seqV2 + +
Version 2.0 (use this one) + +## Version 2.0 (use this one) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | +| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | +| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes```, ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, this is the area of the FOV (aka ROI) region being captured. For Xenium this is the total area of the FOV regions (aka ROI) being captured. For Stereo-Seq this is the number of beads. | | True | +| mapped_area_unit | Assigned Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2```, ```mm^2``` | True | +| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | +| spot_size_unit | Assigned Value | The unit for spot size value. | ```um^2```, ```mm^2``` | True | +| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | +| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | True | +| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | True | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| number_of_additional_stains | Numeric | This would be minimally 2 (always include DAPI and polyT) and can include 6 more. | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/RNAseq.md b/docs/assays/metadata/RNAseq.md index 0592557c..8a07abc0 100644 --- a/docs/assays/metadata/RNAseq.md +++ b/docs/assays/metadata/RNAseq.md @@ -1,95 +1,425 @@ ---- -layout: page-triary ---- - -# RNAseq Metadata Attributes - -Fields that are collected for RNAseq data, available at ```dataset.metadata.``` -  - -* indicates a required field - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| parent_sample_id | | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | -| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | -| preparation_protocol_doi | | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | ```https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1``` | -| dataset_type | | The specific type of dataset being produced. | | -| analyte_class | | Analytes are the target molecules being measured with the assay. | | -| is_targeted | | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | -| acquisition_instrument_vendor | | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | -| acquisition_instrument_model | | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | -| source_storage_duration_value | | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | -| source_storage_duration_unit | | The time duration unit of measurement | | -| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | -| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | | -| contributors_path | | Relative path to file with ORCID IDs for contributors for this dataset. | | -| data_path | | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | -| barcode_offset | | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | | -| barcode_read | | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | | -| barcode_size | | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | | -| umi_offset | | Position in the read at which the umi barcode starts. | | -| umi_read | | Which read file(s) contains the UMI (unique molecular identifier) barcode. | | -| umi_size | | Length of the umi barcode in base pairs. | | -| assay_input_entity | | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | | -| number_of_input_cells_or_nuclei | | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | -| amount_of_input_analyte_value | | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | -| amount_of_input_analyte_unit | | Units of amount of entity input to assay value | | -| library_adapter_sequence | | Adapter sequence to be used for adapter trimming | | -| library_average_fragment_size | | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | -| library_input_amount_value | | The amount of cDNA, after amplification, that was used for library construction. | | -| library_input_amount_unit | | unit of library input amount value | | -| library_output_amount_value | | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | -| library_output_amount_unit | | Units of library final yield. | | -| library_concentration_value | | The concentration value of the pooled library samples submitted for sequencing. | | -| library_concentration_unit | | Unit of library concentration value. | | -| library_layout | | State whether the library was generated for single-end or paired end sequencing. | | -| number_of_iterations_of_cdna_amplification | | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | -| number_of_pcr_cycles_for_indexing | | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | -| library_preparation_kit | | Reagent kit used for library preparation | | -| sample_indexing_kit | | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | | -| sample_indexing_set | | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | -| is_technical_replicate | | Is the sequencing reaction run in replicate, TRUE or FALSE | | -| expected_entity_capture_count | | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | -| sequencing_reagent_kit | | Reagent kit used for sequencing | | -| sequencing_read_format | | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | -| sequencing_batch_id | | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | -| capture_batch_id | | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | -| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | | -| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | -| preparation_instrument_kit | | The reagent kit used with the preparation instrument. | | -| metadata_schema_id | | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | - - -  - -## Deprecated Attributes -  - - indicates a field that was previously required - -| Attribute | Type | Description | Allowable Values | -|------|------|-------------|-------------------| -| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | | -| bulk_rna_isolation_protocols_io_doi | | Link to a protocols document answering the question: How was tissue stored and processed for RNA isolation RNA_isolation_protocols_io_doi | | -| bulk_rna_isolation_quality_metric_value | | RIN value | | -| bulk_rna_yield_units_per_tissue_unit | | RNA amount per Tissue input amount. Valid values should be weight/weight (ng/mg). | | -| bulk_rna_yield_value | | RNA (ng) per Weight of Tissue (mg). Answer the question: How much RNA in ng was isolated? How much tissue in mg was initially used for isolating RNA? Calculate the yield by dividing total RNA isolated by amount of tissue used to isolate RNA from (ng/mg). | | -| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | -| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | -| library_construction_protocols_io_doi | | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | -| library_id | | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | -| operator | | Name of the person responsible for executing the assay. | | -| operator_email | | Email address for the operator. | | -| pi | | Name of the principal investigator responsible for the data. | | -| pi_email | | Email address for the principal investigator. | | -| rnaseq_assay_method | | The kit used for the RNA sequencing assay | | -| sc_isolation_enrichment | | The method by which specific cell populations are sorted or enriched. | | -| sc_isolation_protocols_io_doi | | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | -| sc_isolation_quality_metric | | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | -| sc_isolation_tissue_dissociation | | The method by which tissues are dissociated into single cells in suspension. | | -| sc_isolation_cell_number | | Total number of cell/nuclei yielded post dissociation and enrichment | | -| sequencing_phix_percent | | Percent PhiX loaded to the run | | -| sequencing_read_percent_q30 | | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | -| version | | Version of the schema to use when validating this metadata. | | -| description | | Free-text description of this assay. | | +--- +layout: page +--- +# RNAseq + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
RNAseq Version 5 (current) + +## RNAseq Version 5 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | True | +| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | +| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | +| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | +| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | +| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | +| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | True | +| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | True | +| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | +| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | +| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | +| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | +| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | +| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | +| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | +| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | +| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | +| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | +| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | +| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | +| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | +| number_of_iterations_of_cdna_amplification | Numeric | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | True | +| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | True | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,78``` ```10,48,86``` ```Not applicable``` | True | +| amount_of_input_analyte_unit | Allowable Value | Units of amount of entity input to assay value | ```ug``` ```ng``` | False | + +
+ +
RNAseq Version 2 + +## RNAseq Version 2 + +| Attribute | Type | Description | Allowable Values | required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | True | +| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | +| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | +| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | +| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | +| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | +| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | True | +| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | True | +| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | +| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | +| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | +| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | +| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | +| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | +| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | +| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | +| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | +| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | +| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | +| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | +| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | +| amount_of_input_analyte_unit | Textfield | Units of amount of entity input to assay value | | False | +| number_of_iterations_of_cdna_amplification | Numeric | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | True | +| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,78``` ```10,48,86``` ```Not applicable``` | True | +| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | True | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq (bulk)``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```PhenoCycler``` ```RNAseq (bulk)``` ```scATACseq``` ```scRNAseq``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```snATACseq``` ```snRNAseq``` ```Thick section Multiphoton MxIF``` ```Visium``` ```Xenium``` | True | + +
+ +
bulk-RNA Version 1 + +## bulk-RNA Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['bulkATACseq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| bulk_transposition_input_number_nuclei | Textfield | A number (no comma separators) | | True | +| bulk_atac_cell_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | True | +| is_technical_replicate | Allowable Value | Is this a sequencing replicate? | ['Yes','No']] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | +| library_concentration_unit | Allowable Value | Unit of library_concentration_value | ['nM'] | False | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_creation_date | Datetime | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | False | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_pcr_cycles | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | True | +| library_preparation_kit | Textfield | Reagent kit used for library preparation | | True | +| sample_quality_metric | Textfield | This is a quality metric by visual inspection. This should answerthe question: Are the nuclei intact and are the nuclei free of significant amountsof debris? This can be captured at a high level, “OK” or “notOK”. | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | +| transposition_method | Textfield | Modality of capturing accessible chromatin molecules. The kit used, for example. | | True | +| transposition_transposase_source | Textfield | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
bulk-RNA Version 0 + +## bulk-RNA Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['bulk-RNA'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| bulk_rna_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How was tissue stored and processed for RNA isolation RNA_isolation_protocols_io_doi | | True | +| bulk_rna_yield_value | Numeric | RNA (ng) per Weight of Tissue (mg). Answer the question: How much RNA in ng was isolated? How much tissue in mg was initially used for isolating RNA? Calculate the yield by dividing total RNA isolated by amount of tissue used to isolate RNA from (ng/mg). | | True | +| bulk_rna_yield_units_per_tissue_unit | Allowable Value | RNA amount per Tissue input amount. Valid values should be weight/weight (ng/mg). | ['ng/mg'] | True | +| bulk_rna_isolation_quality_metric_value | Numeric | RIN value | | True | +| rnaseq_assay_input_value | Numeric | RNA input amount value to the assay | | True | +| rnaseq_assay_input_unit | Allowable Value | Units of RNA input amount to the assay | ['ug'] | False | +| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming. | | True | +| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
scRNAseq Version 3 + +## scRNAseq Version 3 + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['3'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The UMI sequence length in the 10xGenomics-v2 kit is 10 base pairs and the length in the 10xGenomics-v3 kit is 12 base pairs. | ['scRNAseq-10xGenomics-v2', 'scRNAseq-10xGenomics-v3', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | +| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | +| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | +| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | +| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | +| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | +| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | +| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, TRUE or FALSE | ['Yes','No']] | True | +| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). | | False | +| umi_read | Textfield | Which read file(s) contains the UMI (unique molecular identifier) barcode. | | True | +| umi_offset | Numeric | Position in the read at which the umi barcode starts. | | True | +| umi_size | Numeric | Length of the umi barcode in base pairs. | | True | +| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | False | +| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | False | +| expected_cell_count | Numeric | How many cells are expected? This may be used in downstream pipelines to guide selection of cell barcodes or segmentation parameters. | | False | +| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | +| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
scRNAseq Version 2 + +## scRNAseq Version 2 + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics-v2', 'scRNAseq-10xGenomics-v3', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | +| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | +| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | +| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | +| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | +| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | +| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | +| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, TRUE or FALSE | ['Yes','No']] | True | +| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). | | False | +| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | False | +| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | False | +| expected_cell_count | Numeric | How many cells are expected? This may be used in downstream pipelines to guide selection of cell barcodes or segmentation parameters. | | False | +| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | +| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
scRNAseq Version 1 + +## scRNAseq Version 1 + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | +| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | +| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | +| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | +| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | +| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | +| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | +| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in repliucate, TRUE or FALSE | ['Yes','No']] | True | +| cell_barcode_read | Textfield | Which read file contains the cell barcode | | True | +| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | True | +| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | True | +| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | +| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
scRNAseq Version 0 + +## scRNAseq Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics-v2', 'scRNAseq-10xGenomics-v3', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | +| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | +| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | +| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | +| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | +| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | +| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | +| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, TRUE or FALSE | ['Yes','No']] | True | +| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). | | False | +| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | False | +| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | False | +| expected_cell_count | Numeric | How many cells are expected? This may be used in downstream pipelines to guide selection of cell barcodes or segmentation parameters. | | False | +| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | +| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/RNAseqWithProbes.md b/docs/assays/metadata/RNAseqWithProbes.md new file mode 100644 index 00000000..fb8e9742 --- /dev/null +++ b/docs/assays/metadata/RNAseqWithProbes.md @@ -0,0 +1,63 @@ +--- +layout: page +--- +# RNAseq-(with-probes) +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```8,8``` ```Not applicable``` | True | +| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | +| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```14``` ```Not applicable``` | True | +| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | +| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | +| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | +| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | +| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | False | +| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | False | +| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | +| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | +| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | +| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | +| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | +| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | +| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | False | +| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| False | +| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | False | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | +| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | +| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | +| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | +| number_of_iterations_of_cdna_amplification | Numeric | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | True | +| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```34``` ```36``` ```Not applicable``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | True | +| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ```Yes``` ```No``` | True | +| probe_hybridization_time_value | Numeric | How long was the oligo-conjugated RNA or oligo-conjugated antibody probes hybridized with the sample? | | True | +| probe_hybridization_time_unit | Allowable Value | The units for probe hybridization time value. | ```Hour``` ```Minute``` | True | +| oligo_probe_panel | Allowable Value | This is the probe panel used to target genes and/or proteins. In cases where there is a core panel and add-on modules, the core panel should be selected here. If additional panels are used, then they must be included in the "additional_panels_used.csv" file that's uploaded with the dataset. | ```10x Genomics; Chromium Fixed RNA Kit``` ```Human Transcriptome``` ```4 rxns x 1 BC; PN 1000474``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```16 rxns; PN 1000420``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```64 rxns; PN 1000456``` ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363``` ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365``` ```Custom``` ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-HuWTA-4``` ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-MsWTA-4``` | True | +| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ```Yes``` ```No``` | True | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| amount_of_input_analyte_unit | Allowable Value | Units of amount of entity input to assay value | ```ug``` ```ng``` | False | + +
diff --git a/docs/assays/metadata/Raman-Imaging.md b/docs/assays/metadata/Raman-Imaging.md new file mode 100644 index 00000000..95665aef --- /dev/null +++ b/docs/assays/metadata/Raman-Imaging.md @@ -0,0 +1,45 @@ +--- +layout: page +--- +# Raman-Imaging + +
Version 2 (current) + +## Version 2 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| is_image_preprocessing_required | Radio | Indicates whether image preprocessing is necessary based on the type of acquisition instrument used, such as a microscope or slide scanner. This may involve steps like fusing image tiles to assemble the complete image. Example: Yes | ```Yes```, ```No``` | False | +| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | False | +| tiled_image_columns | Numeric | The number of columns used in the stitching process of a tiled image, often referred to as the grid size in the x-dimension. Example: 5 | | False | +| tiled_image_count | Numeric | The total number of raw tiled images captured, which are intended to be stitched together. Example: 75 | | False | +| intended_tile_overlap_percentage | Numeric | The intended percentage of overlap between tiled images. This value serves as the set point, although slight variations may occur during image acquisition due to stage registration. Example: 5 | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq```, ```PhenoCycler``` | True | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | +| tile_configuration | Assigned Value | The configuration of tiles used for stitching in the assay process. If no tile configuration is applicable, enter "Not applicable". Example: Row-by-row | ```Column-by-column```, ```Not applicable```, ```Snake-by-columns```, ```Row-by-row```, ```Snake-by-rows``` | False | +| scan_direction | Assigned Value | The direction of imaging, which is necessary for the stitching process. Example: Left-and-down | ```Left-and-down```, ```Right-and-down```, ```Not applicable```, ```Right-and-up```, ```Left-and-up``` | False | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| number_of_pixels | Numeric | The total number of spatial sampling points in an image; for example, in a Raman image, each pixel corresponds to one recorded Raman spectrum. Example: 40000 | | True | +| pixel_physical_size_height_value | Numeric | The physical height of a single pixel in the image. Example: 1000 | | True | +| pixel_physical_size_height_unit | Assigned Value | The unit of measurement for the pixel physical size height value. If the pixel height is not specified, this field may be left blank. Example: um | ```um```, ```mm```, ```nm``` | True | +| pixel_physical_size_width_value | Numeric | The physical width of a single pixel in the image. Example: 1000 | | True | +| pixel_physical_size_width_unit | Assigned Value | The unit of measurement for the pixel physical size width value. If the pixel width value is not specified, this field may be left blank. Example: um | ```um```, ```mm```, ```nm``` | True | +| pixel_physical_size_depth_value | Numeric | The physical depth of a single pixel in the image. Example: 10 | | True | +| pixel_physical_size_depth_unit | Assigned Value | The unit of measurement for the pixel physical size depth value. If the pixel depth value is not specified, this field may be left blank. Example: um | ```um```, ```mm```, ```nm``` | True | +| objective_numerical_aperture | Numeric | Numerical aperture of the microscope objective used to focus the excitation laser on the sample and collect the resulting scattered signal, such as Raman-scattered light. Example: 0.5 | | True | +| laser_power | Numeric | Power of the excitation laser at the sample’s focal plane, measured after the objective and reported in milliwatts (mW). Example: 10 | | True | +| raman_shift_range | Textfield | Range of Raman shifts acquired in the measurement, expressed in wavenumbers (cm⁻¹). Example: 400-3200 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/SIMS.md b/docs/assays/metadata/SIMS.md new file mode 100644 index 00000000..3880a74d --- /dev/null +++ b/docs/assays/metadata/SIMS.md @@ -0,0 +1,39 @@ +--- +layout: page +--- +# SIMS + +
Version 2 (latest) + +## Version 2 (latest) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | +| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | True | +| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | +| matrix_deposition_method | Allowable Value | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | False | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | +| preparation_matrix | Allowable Value | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | +| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | +| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | +| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | +| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
diff --git a/docs/assays/metadata/STARmap.md b/docs/assays/metadata/STARmap.md new file mode 100644 index 00000000..030cdade --- /dev/null +++ b/docs/assays/metadata/STARmap.md @@ -0,0 +1,43 @@ +--- +layout: page +--- +# STARmap + +
Version 2.0 (use this one) + +## Version 2.0 (use this one) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq```, ```PhenoCycler``` | True | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | True | +| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | True | +| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | +| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | +| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | +| proteinasek_concentration | Numeric | The concentration of the enzyme Proteinase K within a sample, measured in micrograms per milliliter (ug/ml). Example: 10 | | False | +| proteinasek_incubation_time_value | Numeric | The duration for which a sample is incubated with Proteinase K. Example: 15 | | False | +| proteinasek_incubation_time_unit | Assigned Value | The unit of measurement for the proteinaseK incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | +| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | +| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | +| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | +| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | +| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | +| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | +| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/SecondHarmonicGeneration.md b/docs/assays/metadata/SecondHarmonicGeneration.md new file mode 100644 index 00000000..22b4169b --- /dev/null +++ b/docs/assays/metadata/SecondHarmonicGeneration.md @@ -0,0 +1,34 @@ +--- +layout: page +--- +# SGH +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | +| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | +| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
diff --git a/docs/assays/metadata/Seq-Scope.md b/docs/assays/metadata/Seq-Scope.md new file mode 100644 index 00000000..e8e33049 --- /dev/null +++ b/docs/assays/metadata/Seq-Scope.md @@ -0,0 +1,39 @@ +--- +layout: page +--- +# Seq-Scope + +
Version 2 (current) + +## Version 2 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | +| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | +| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | +| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | +| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | +| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | +| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, this is the area of the FOV (aka ROI) region being captured. For Xenium this is the total area of the FOV regions (aka ROI) being captured. For Stereo-Seq this is the number of beads. | | True | +| mapped_area_unit | Assigned Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2```, ```mm^2``` | True | +| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | +| spot_size_unit | Assigned Value | The unit for spot size value. | ```um^2```, ```mm^2``` | True | +| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | +| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | +| spot_spacing_unit | Assigned Value | Units corresponding to inter-spot distance | ```um``` | True | +| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | +| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | False | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| number_of_additional_stains | Numeric | This would be minimally 2 (always include DAPI and polyT) and can include 6 more. | | False | + +
\ No newline at end of file diff --git a/docs/assays/metadata/Slide-seq.md b/docs/assays/metadata/Slide-seq.md new file mode 100644 index 00000000..3f7fad31 --- /dev/null +++ b/docs/assays/metadata/Slide-seq.md @@ -0,0 +1,94 @@ +--- +layout: page +--- +# SnareSeq2 + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
Version 1 (no longer accepting data) + +## Version 1 (no longer accepting data) + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['Slide-seq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes', 'No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| puck_id | Textfield | Slide-seq captures RNA sequence data on spatially barcoded arrays of beads. Beads are fixed to a slide in a region shaped like a round puck. Each puck has a unique puck_id. | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in repliucate, TRUE or FALSE | ['Yes', 'No'] | True | +| bead_barcode_read | Textfield | Which read file contains the bead barcode | | True | +| bead_barcode_offset | Textfield | Position(s) in the read at which the bead barcode starts | | True | +| bead_barcode_size | Textfield | Length of the bead barcode in base pairs | | True | +| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | +| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
Version 0 + +## Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['Slide-seq'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes', 'No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | +| puck_id | Textfield | Slide-seq captures RNA sequence data on spatially barcoded arrays of beads. Beads are fixed to a slide in a region shaped like a round puck. Each puck has a unique puck_id. | | True | +| is_technical_replicate | Allowable Value | Is the sequencing reaction run in repliucate, TRUE or FALSE | ['Yes', 'No'] | True | +| bead_barcode_read | Textfield | Which read file contains the bead barcode | | True | +| bead_barcode_offset | Textfield | Position(s) in the read at which the bead barcode starts | | True | +| bead_barcode_size | Textfield | Length of the bead barcode in base pairs | | True | +| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | +| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | +| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | +| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
diff --git a/docs/assays/metadata/SnareSeq2.md b/docs/assays/metadata/SnareSeq2.md new file mode 100644 index 00000000..2ce1a047 --- /dev/null +++ b/docs/assays/metadata/SnareSeq2.md @@ -0,0 +1,19 @@ +--- +layout: page +--- +# SnareSeq2 +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|----------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| number_of_pre-amplification_pcr_cycles | Numeric | The number of PCR cycles run after the Chromium Controller step and prior to separating the suspension and initiating library construction | | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/ThickSectionMultiphotonMxIF.md b/docs/assays/metadata/ThickSectionMultiphotonMxIF.md new file mode 100644 index 00000000..a6238240 --- /dev/null +++ b/docs/assays/metadata/ThickSectionMultiphotonMxIF.md @@ -0,0 +1,34 @@ +--- +layout: page +--- +# MxIF +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | +| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | +| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | +| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | +| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | +| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | +| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | +| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | +| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | +| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
diff --git a/docs/assays/metadata/Visium-HD.md b/docs/assays/metadata/Visium-HD.md new file mode 100644 index 00000000..c1cae516 --- /dev/null +++ b/docs/assays/metadata/Visium-HD.md @@ -0,0 +1,33 @@ +--- +layout: page +--- +# Visium-HD + +
Version 2 (current) + +## Version 2 (current) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | +| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | +| dataset_type | Assigned Value | The specific type of dataset being produced. | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, this is the area of the FOV (aka ROI) region being captured. For Xenium this is the total area of the FOV regions (aka ROI) being captured. For Stereo-Seq this is the number of beads. | | True | +| mapped_area_unit | Assigned Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2```, ```mm^2``` | True | +| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | +| spot_size_unit | Assigned Value | The unit for spot size value. | ```um^2```, ```mm^2``` | True | +| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | +| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | +| spot_spacing_unit | Assigned Value | Units corresponding to inter-spot distance | ```um``` | True | +| capture_area_id | Radio | Which capture area on the slide was used. For Visium this would be [A1, B1, C1, D1]. For HiFi this would be the lane on the flowcell. | ```A1```, ```B1```, ```C1```, ```D1```, ```Lane 1```, ```Lane 2```, ```Lane 3```, ```Lane 4```, ```Lane 5```, ```Lane 6```, ```Lane 7```, ```Lane 8``` | True | +| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | +| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| preparation_instrument_vendor | Assigned Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```Thermo Fisher Scientific```, ```SunChrom```, ```Leica Biosystems```, ```Roche Diagnostics```, ```In-House```, ```Not applicable```, ```Hamamatsu```, ```HTX Technologies```, ```10x Genomics``` | True | +| preparation_instrument_model | Assigned Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL```, ```ST5020 Multistainer```, ```Visium CytAssist```, ```SunCollect Sprayer```, ```Chromium X```, ```Chromium iX```, ```EVOS M7000```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```Discovery Ultra```, ```Sublimator```, ```Not applicable```, ```TM-Sprayer```, ```M5 Sprayer```, ```M3+ Sprayer```, ```Chromium Controller```, ```Chromium Connect``` | True | +| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | + +
\ No newline at end of file diff --git a/docs/assays/metadata/VisiumNoProbes.md b/docs/assays/metadata/VisiumNoProbes.md new file mode 100644 index 00000000..2d741c42 --- /dev/null +++ b/docs/assays/metadata/VisiumNoProbes.md @@ -0,0 +1,58 @@ +--- +layout: page +--- +# Visium-(no-probes) + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
Version3 (current) + +## Version 3 + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | +| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | True | +| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | +| spot_size_unit | Allowable Value | The unit for spot size value. | ```um^2``` ```mm^2``` | True | +| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | +| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | +| spot_spacing_unit | Allowable Value | Units corresponding to inter-spot distance | ```um``` | True | +| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be ```A1, B1, C1, D1```. For HiFi this would be the lane on the flowcell. | ```A1``` ```B1``` ```C1``` ```D1``` ```Lane 1``` ```Lane 2``` ```Lane 3``` ```Lane 4``` ```Lane 5``` ```Lane 6``` ```Lane 7``` ```Lane 8``` | True | +| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | +| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | + +
+ +
Version 2 + +## Version 2 + +| Attribute | Type | Description | Allowable Value | Required | +|-----------------------------|----------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------|------------| +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| dataset_type | Textfield | The specific type of dataset being produced. | | True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | +| mapped_area_unit | Textfield | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | | True | +| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | +| spot_size_unit | Textfield | The unit for spot size value. | | True | +| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | +| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | +| spot_spacing_unit | Textfield | Units corresponding to inter-spot distance | | True | +| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be [A1, B1, C1, D1]. For HiFi this would be the lane on the flowcell. | [A1, B1, C1, D1, Lane 1, Lane 2, Lane 3, Lane 4, Lane 5, Lane 6, Lane 7, Lane 8] | True | +| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | +| permeabilization_time_unit | Textfield | The unit for the permeabilization time. | | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/VisiumWithProbes.md b/docs/assays/metadata/VisiumWithProbes.md new file mode 100644 index 00000000..89c07f31 --- /dev/null +++ b/docs/assays/metadata/VisiumWithProbes.md @@ -0,0 +1,30 @@ +--- +layout: page +--- +# Visium-(with-probes) +
Version 2 (current) + +## Version 2 (current) + +| Attribute | Type | Description | Allowable Values | Required | +|-------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| +| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | +| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | +| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | +| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | +| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | True | +| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | +| spot_size_unit | Allowable Value | The unit for spot size value. | ```um^2``` ```mm^2``` | True | +| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | +| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | +| spot_spacing_unit | Allowable Value | Units corresponding to inter-spot distance | ```um``` | True | +| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be ```A1, B1, C1, D1```. For HiFi this would be the lane on the flowcell. | ```A1``` ```B1``` ```C1``` ```D1``` ```Lane 1``` ```Lane 2``` ```Lane 3``` ```Lane 4``` ```Lane 5``` ```Lane 6``` ```Lane 7``` ```Lane 8``` | True | +| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | +| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | False | +| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | +| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | True | +| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| True | + +
diff --git a/docs/assays/metadata/WGS.md b/docs/assays/metadata/WGS.md new file mode 100644 index 00000000..20a9ee8b --- /dev/null +++ b/docs/assays/metadata/WGS.md @@ -0,0 +1,86 @@ +--- +layout: page +--- +# WGS + +NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. + +
Version 1 (no longer accepting data) + +## Version 1 (no longer accepting data) + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| +| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | +| description | Textfield | Free-text description of this assay. | | True | +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['WGS'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| gdna_fragmentation_quality_assurance | Allowable Value | Is the gDNA integrity good enough for WGS? This is usually checked through running a gel. | ['Pass', 'Fail'] | True | +| dna_assay_input_value | Numeric | Amount of DNA input into library preparation | | True | +| dna_assay_input_unit | Allowable Value | Units of DNA input into library preparation | ['ug'] | False | +| library_construction_method | Textfield | Describes DNA library preparation kit. Modality of isolating gDNA, Fragmentation and generating sequencing libraries. | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used. | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | The adapter sequence to be used for adapter trimming starting with the 5' end. (eg. 5-ATCCTGAGAA) | | True | +| library_final_yield | Numeric | Total amount of library after final pcr amplification step | | True | +| library_final_yield_unit | Allowable Value | Total units of library after final pcr amplification step | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
+ +
Version 0 + +## Version 0 + +| Attribute | Type | Description | Allowable Values | Required | +|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| +| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | +| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | +| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | +| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | +| operator | Textfield | Name of the person responsible for executing the assay. | | True | +| operator_email | Textfield | Email address for the operator. | | True | +| pi | Textfield | Name of the principal investigator responsible for the data. | | True | +| pi_email | Textfield | Email address for the principal investigator. | | True | +| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | +| assay_type | Allowable Value | The specific type of assay being executed. | ['WGS'] | True | +| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | +| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | +| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | +| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | +| gdna_fragmentation_quality_assurance | Allowable Value | Is the gDNA integrity good enough for WGS? This is usually checked through running a gel. | ['Pass', 'Fail'] | True | +| dna_assay_input_value | Numeric | Amount of DNA input into library preparation | | True | +| dna_assay_input_unit | Allowable Value | Units of DNA input into library preparation | ['ug'] | False | +| library_construction_method | Textfield | Describes DNA library preparation kit. Modality of isolating gDNA, Fragmentation and generating sequencing libraries. | | True | +| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used. | | True | +| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | +| library_adapter_sequence | Textfield | The adapter sequence to be used for adapter trimming starting with the 5' end. (eg. 5-ATCCTGAGAA) | | True | +| library_final_yield | Numeric | Total amount of library after final pcr amplification step | | True | +| library_final_yield_unit | Allowable Value | Total units of library after final pcr amplification step | ['ng'] | False | +| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | +| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | +| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | +| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | +| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | +| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | +| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | + +
diff --git a/docs/assays/metadata/iCLAP.md b/docs/assays/metadata/iCLAP.md new file mode 100644 index 00000000..ce13e526 --- /dev/null +++ b/docs/assays/metadata/iCLAP.md @@ -0,0 +1,34 @@ +--- +layout: page +--- +# iCLAP + +
Version 2.0 (use this one) + +## Version 2.0 (use this one) + +| Attribute Name | Type | Description | Allowable Values | Required | +|---------------|------|-------------|------------------|----------| +| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | +| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | +| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | +| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | +| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | +| number_of_antibodies | Numeric | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | | True | +| number_of_biomarker_imaging_rounds | Numeric | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | | True | +| number_of_total_imaging_rounds | Numeric | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | | True | +| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | +| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq``` | True | +| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | +| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | +| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | +| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | +| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | +| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | +| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | +| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | +| antibodies_path | Textfield | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | | True | +| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | +| number_of_channels | Numeric | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | True | + +
\ No newline at end of file diff --git a/docs/assays/metadata/index.md b/docs/assays/metadata/index.md index efa18772..ddda0ecb 100644 --- a/docs/assays/metadata/index.md +++ b/docs/assays/metadata/index.md @@ -3,7 +3,8 @@ layout: page --- ## HuBMAP Metadata by Dataset Type -A list of available dataset types (data types from multiple supported assays), with a link [](EnhancedSRS "Attribute description") to the valid metadata attributes for each dataset type. The linked assay metadata pages list all attributes, as they have occurred, across any versions of the metadata specification for the given dataset type with the most current, valid set of attributes listed first on the page. The directory schema for each dataset type is also linked in the description column. +A list of available dataset types (data types from multiple supported assays), with a link [](EnhancedSRS "Attribute description") to the valid metadata attributes for each dataset type. The linked assay metadata pages list all attributes, as they have occurred, across any versions of the metadata specification for the given dataset type with the most current, valid set of attributes listed first on the page. The directory schema for each dataset type is also linked in the description column. + | Dataset Type | Description | |--------------|-------------| @@ -25,24 +26,24 @@ A list of available dataset types (data types from multiple supported assays), w | [IMC](https://docs.hubmapconsortium.org/assays/imc) [](IMC "Attribute description")| Combines standard immunohistochemistry with CyTOF mass cytometry to resolve the cellular localization of up to 40 proteins in a tissue sample. Link to [IMC directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/imc-2d/current/). | | [LC-MS](https://docs.hubmapconsortium.org/assays/lcms) [](LC-MS "Attribute description")| Coupling of liquid chromatography (LC) to mass spectrometry (MS). Link to [LC-MS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/lcms/current/). | | [Light Sheet](https://en.wikipedia.org/wiki/Light_sheet_fluorescence_microscopy) [](LightSheet "Attribute description")| A fluorescence imaging technique that uses a thin sheet of laser light to illuminate a sample, allowing for high-resolution, 3D imaging with reduced photobleaching and phototoxicity; particularly useful for imaging large, thick, or delicate biological samples, like developing embryos or organoids. Link to [Light Sheet directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/lightsheet/current/). | -| [MALDI-IMS](https://docs.hubmapconsortium.org/assays/maldi-ims) [](MALDI "Attribute description") | Matrix-assisted laser desorption/ionization (MALDI) imaging mass spectrometry (IMS) combines the sensitivity and molecular specificity of MS with the spatial fidelity of classical microscopy. Link to [MALDI-IMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/maldi/current/). | -| [MIBI](https://www.researchgate.net/figure/Multiplexed-ion-beam-imaging-workflow-for-high-resolution-spatial-proteomics-Here_fig1_349770840) [](MIBI "Attribute description") | Preserved tissue sections, mounted on conductive substrates are incubated with unique isotopic transition metal-tagged antibody reporters. An oxygen primary ion beam rasters the sample surface, ejecting and ionizing the isotope reporters. Their masses are subsequently measured via a mass analyzer. Link to [MIBI directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/mibi/current/). | -| [MERFISH](https://pubmed.ncbi.nlm.nih.gov/27241748/) [](MERFISH "Attribute description") | A spatial transcriptomics technology that allows for the simultaneous imaging of hundreds to thousands of RNA species within single cells, providing both copy number and spatial distribution information. Link to [MERFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/merfish/current/). | -| [MUSIC](https://www.nature.com/articles/s41586-024-07239-w) [](MUSIC "Attribute description") | A sequencing assay that allows profiling of gene expression, co-complexed DNA sequences, and RNA-chromatin interactions from the same single-cell nucleus. Both RNA and fragmented DNA within a nucleus are labelled with a unique cell barcode, enabling identification and matching of RNA and DNA sequences. Link to [MUSIC directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/music/current/). | -| [MxIF](https://pmc.ncbi.nlm.nih.gov/articles/PMC9959383/#) [](ThickSectionMultiphotonMxIF "Attribute description") | One version of MXIF (multiplexed fluorescence microscopy), an imaging platform whereby a large number of cellular and histological markers can be investigated on a single tissue section. Link to [TSM MxIF directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/thick-section-multiphoton-mxif/current/).| -| Pixel-seqV2 [](Pixel-seqV2 "Attribute description") | Pixel-seqV2 is a spatial transcriptomics method that utilizes polony gels to capture and sequence RNA, proteins or other molecules in tissues with high resolution. These polony gels are arrays of micron-scale DNA clusters, each containing a unique barcode, allowing for the mapping of molecules within their original spatial context in a tissue, thereby allowing researchers to study the spatial organization of cells and their gene expression profiles within tissues.| -| [RNAseq](https://docs.hubmapconsortium.org/assays/rnaseq) [](RNAseq "Attribute description") | While bulk RNAseq elucidates the average gene expression profile in cells comprising a tissue sample, single-cell RNAseq employs per-cell and per-molecule barcoding to enable single-cell resolution of the gene expression profile. Link to [RNAseq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq/current/).| -| [RNAseq with Probes](https://pmc.ncbi.nlm.nih.gov/articles/PMC5717752/#) [](RNAseqWithProbes "Attribute description") | Uses probes to capture and enrich specific regions of the RNA for targeted sequencing, allowing for in-depth analysis of those regions. Link to [RNAseq with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq-with-probes/current/).| -| Raman-Imaging [](Raman-Imaging "Attribute description") | Raman Imaging is a non-invasive technique that maps the unique chemical fingerprint of biological samples (cells, tissues) by capturing Raman scattering (light interacting with molecular vibrations) at each pixel, creating detailed molecular maps showing the distribution of proteins, lipids, DNA, and water. Link to [Raman Imaging directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/raman-imaging/current/). | -| [SHG](https://en.wikipedia.org/wiki/Second-harmonic_imaging_microscopy) [](SecondHarmonicGeneration "Attribute description") | Single-cycle Fluorescence Microscopy (SFM). A technique that utilizes the nonlinear optical phenomenon of SHG to image biological tissues and structures, particularly those containing collagen. Link to [SHG directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/second-harmonic-generation/current/).| -| [SeqFISH](https://docs.hubmapconsortium.org/assays/seqfish) [](seqFISH "Attribute description") | SeqFISH technology allows in situ imaging of multiple mRNAs using barcoding and fluorophore-labelled barcode readout-probes. Link to [SeqFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/seqfish/). _The consortium is no longer accepting data of this type_.| -| [SIMS](https://www.frontiersin.org/journals/chemistry/articles/10.3389/fchem.2023.1237408/full) [](SIMS "Attribute description") | Secondary-ion mass spectrometry (SIMS) is a technique used to analyze the composition of solid surfaces and thin films by sputtering the surface of the specimen. Link to [SIMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/sims/current/ "directory schema").| -| [Slide-seq](https://www.nature.com/articles/s41587-020-0739-1) [](Slide-seq "Attribute description") | Provides a scalable method for obtaining spatially resolved gene expression data at resolutions comparable to the sizes of individual cells. Link to [Slide-seq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/slide-seq/current/).| -| [SnareSeq2](https://www.nature.com/articles/s41596-021-00507-3) [](SnareSeq2 "Attribute description") | This method uses tagmentation within permeabilized and fixed single-nucleus isolates to capture accessible chromatin (AC) regions, followed by the capture and reverse transcription of RNA transcripts. Link to [SnareSeq2 directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/snareseq2/current/).| -| STARmap [](STARmap "Attribute description") | STARmap (Spatially-resolved Transcript Amplicon Readout Mapping) is a biomedical technology that enables the 3D mapping of gene expression within intact tissues at single-cell resolution. It combines hydrogel-tissue chemistry and in situ DNA sequencing to preserve a cell's location and identify which genes are active in that specific spatial context. [STARmap directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/starmap/current/). | -| [Visium No Probes](https://ostr.ccr.cancer.gov/emerging-technologies/spatial-biology/visium/) [](VisiumNoProbes "Attribute description") | A spatial transcriptomics solution that allows researchers to analyze gene expression patterns within the spatial context of a tissue. An in situ method that captures RNA transcripts within the tissue and then sequences them. Link to [Visium NP directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-no-probes/current/). | -| [Visium with Probes](https://ngisweden.scilifelab.se/methods/10x-genomics-visium-cytassist-for-ffpe-samples/) [](VisiumWithProbes "Attribute description") | Offers spatially resolved transcriptomics through the 10X Genomics Visium CytAssist, which combines histology with probe-based transcriptomics in a spatial context. Link to [Visium with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-with-probes/current/). | -| [WGS](https://en.wikipedia.org/wiki/Whole_genome_sequencing) [](WGS "Attribute description") | The process of determining the entire DNA sequence of an organism's genome at a single time. This entails sequencing all of an organism's chromosomal and mitochondrial DNA. Link to [WGS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/wgs/). _The consortium is no longer accepting data of this type_. | +| [MALDI-IMS](https://docs.hubmapconsortium.org/assays/maldi-ims) [](MALDI "Attribute description") | Matrix-assisted laser desorption/ionization (MALDI) imaging mass spectrometry (IMS) combines the sensitivity and molecular specificity of MS with the spatial fidelity of classical microscopy. Link to [MALDI-IMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/maldi/current/). | +| [MIBI](https://www.researchgate.net/figure/Multiplexed-ion-beam-imaging-workflow-for-high-resolution-spatial-proteomics-Here_fig1_349770840) [](MIBI "Attribute description") | Preserved tissue sections, mounted on conductive substrates are incubated with unique isotopic transition metal-tagged antibody reporters. An oxygen primary ion beam rasters the sample surface, ejecting and ionizing the isotope reporters. Their masses are subsequently measured via a mass analyzer. Link to [MIBI directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/mibi/current/). | +| [MERFISH](https://pubmed.ncbi.nlm.nih.gov/27241748/) [](MERFISH "Attribute description") | A spatial transcriptomics technology that allows for the simultaneous imaging of hundreds to thousands of RNA species within single cells, providing both copy number and spatial distribution information. Link to [MERFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/merfish/current/). | +| [MUSIC](https://www.nature.com/articles/s41586-024-07239-w) [](MUSIC "Attribute description") | A sequencing assay that allows profiling of gene expression, co-complexed DNA sequences, and RNA-chromatin interactions from the same single-cell nucleus. Both RNA and fragmented DNA within a nucleus are labelled with a unique cell barcode, enabling identification and matching of RNA and DNA sequences. Link to [MUSIC directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/music/current/). | +| [MxIF](https://pmc.ncbi.nlm.nih.gov/articles/PMC9959383/#) [](ThickSectionMultiphotonMxIF "Attribute description") | One version of MXIF (multiplexed fluorescence microscopy), an imaging platform whereby a large number of cellular and histological markers can be investigated on a single tissue section. Link to [TSM MxIF directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/thick-section-multiphoton-mxif/current/).| +| Pixel-seqV2 [](Pixel-seqV2 "Attribute description") | Pixel-seqV2 is a spatial transcriptomics method that utilizes polony gels to capture and sequence RNA, proteins or other molecules in tissues with high resolution. These polony gels are arrays of micron-scale DNA clusters, each containing a unique barcode, allowing for the mapping of molecules within their original spatial context in a tissue, thereby allowing researchers to study the spatial organization of cells and their gene expression profiles within tissues.| +| [RNAseq](https://docs.hubmapconsortium.org/assays/rnaseq) [](RNAseq "Attribute description") | While bulk RNAseq elucidates the average gene expression profile in cells comprising a tissue sample, single-cell RNAseq employs per-cell and per-molecule barcoding to enable single-cell resolution of the gene expression profile. Link to [RNAseq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq/current/).| +| [RNAseq with Probes](https://pmc.ncbi.nlm.nih.gov/articles/PMC5717752/#) [](RNAseqWithProbes "Attribute description") | Uses probes to capture and enrich specific regions of the RNA for targeted sequencing, allowing for in-depth analysis of those regions. Link to [RNAseq with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq-with-probes/current/).| +| Raman-Imaging [](Raman-Imaging "Attribute description") | Raman Imaging is a non-invasive technique that maps the unique chemical fingerprint of biological samples (cells, tissues) by capturing Raman scattering (light interacting with molecular vibrations) at each pixel, creating detailed molecular maps showing the distribution of proteins, lipids, DNA, and water. Link to [Raman Imaging directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/raman-imaging/current/). | +| [SHG](https://en.wikipedia.org/wiki/Second-harmonic_imaging_microscopy) [](SecondHarmonicGeneration "Attribute description") | Single-cycle Fluorescence Microscopy (SFM). A technique that utilizes the nonlinear optical phenomenon of SHG to image biological tissues and structures, particularly those containing collagen. Link to [SHG directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/second-harmonic-generation/current/).| +| [SeqFISH](https://docs.hubmapconsortium.org/assays/seqfish) [](seqFISH "Attribute description") | SeqFISH technology allows in situ imaging of multiple mRNAs using barcoding and fluorophore-labelled barcode readout-probes. Link to [SeqFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/seqfish/). _The consortium is no longer accepting data of this type_.| +| [SIMS](https://www.frontiersin.org/journals/chemistry/articles/10.3389/fchem.2023.1237408/full) [](SIMS "Attribute description") | Secondary-ion mass spectrometry (SIMS) is a technique used to analyze the composition of solid surfaces and thin films by sputtering the surface of the specimen. Link to [SIMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/sims/current/ "directory schema").| +| [Slide-seq](https://www.nature.com/articles/s41587-020-0739-1) [](Slide-seq "Attribute description") | Provides a scalable method for obtaining spatially resolved gene expression data at resolutions comparable to the sizes of individual cells. Link to [Slide-seq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/slide-seq/current/).| +| [SnareSeq2](https://www.nature.com/articles/s41596-021-00507-3) [](SnareSeq2 "Attribute description") | This method uses tagmentation within permeabilized and fixed single-nucleus isolates to capture accessible chromatin (AC) regions, followed by the capture and reverse transcription of RNA transcripts. Link to [SnareSeq2 directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/snareseq2/current/).| +| STARmap [](STARmap "Attribute description") | STARmap (Spatially-resolved Transcript Amplicon Readout Mapping) is a biomedical technology that enables the 3D mapping of gene expression within intact tissues at single-cell resolution. It combines hydrogel-tissue chemistry and in situ DNA sequencing to preserve a cell's location and identify which genes are active in that specific spatial context. [STARmap directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/starmap/current/). | +| [Visium No Probes](https://ostr.ccr.cancer.gov/emerging-technologies/spatial-biology/visium/) [](VisiumNoProbes "Attribute description") | A spatial transcriptomics solution that allows researchers to analyze gene expression patterns within the spatial context of a tissue. An in situ method that captures RNA transcripts within the tissue and then sequences them. Link to [Visium NP directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-no-probes/current/). | +| [Visium with Probes](https://ngisweden.scilifelab.se/methods/10x-genomics-visium-cytassist-for-ffpe-samples/) [](VisiumWithProbes "Attribute description") | Offers spatially resolved transcriptomics through the 10X Genomics Visium CytAssist, which combines histology with probe-based transcriptomics in a spatial context. Link to [Visium with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-with-probes/current/). | +| [WGS](https://en.wikipedia.org/wiki/Whole_genome_sequencing) [](WGS "Attribute description") | The process of determining the entire DNA sequence of an organism's genome at a single time. This entails sequencing all of an organism's chromosomal and mitochondrial DNA. Link to [WGS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/wgs/). _The consortium is no longer accepting data of this type_. | diff --git a/docs/assays/metadata/testing/info3.png b/docs/assays/metadata/testing/info3.png new file mode 100644 index 0000000000000000000000000000000000000000..811c300e2d8c4f154f3319a208424e3e135518ef GIT binary patch literal 2038 zcmY*ae>~IqAOFl)Hu;exQgXVuJGafukC|U`JqzzH|GoDQrssKvxJFpL^@7w-dRcJ=Jl3W4c+D|&GG%ZyP<#~i80Dz9Z z+CZn5!4;|tIqZO7elW=!%ZQ6b(3o*_7D5=yQ%wT^ZjX>hV;p7iE$FN$HWzRG>Za7% zg3ZKR2RV>zNxc26Xtqa6Jj*}DCxDT1l;O;@-b2uZ;e=QfLM)3MRxqMV(bk+yb7J3F{)0-lh} z<7~wx1)mv5Bzsjg(`d3{RS0$-HrQNqhB{~40wd6^34T#?VvNR;REYPU? z_~P!;JKo;@Vll$x!_+`+MV*IRns0M?xlt&kaZ5PGw?ixFwQI$xlT%sxu$aJc*HiQk zU9G2nu{P!6Tyb7Bb*wTsw&ml(P2RCy@7twDsik{1qMta;K3oXgmm@k%6?HL$Id$sXa;d!ZP82~YyzFQprp0NEM%_Y>MiA}dx4 z=m`)9eg&kFrHt}z*GaN{3Ya3;^W6#+u)EZ8$hEU8-94478M-zPxv(KD@ifB{IWm~x zk`ZF4yzw#}v3xRgbpAu*FJ3ZDDVp?jXYP%U%Z5UOCF~A+H}mU3V5gg9F5CIXp^4y& zE()s%n@0Zy(?1~I5Ui+F#!+C_=K|}gy#xrZP|y5|;Rr+#R-Bd-I`U+lpb-XYdl(Sb z3@qALq0=7Cy=Xpa{GsmTxI80Abn)S6@m~Gc?)Gxwca>}Jhzt&3PA0E+R3nF;(G~iv zQPGSHjz}|a1_fypvte_T_Zj)JBlmlU&%I?McF1xHI3X11>N9WYz=5zrSiZgw#ZW7ej7ZIk4iEeBDg?)FjliKGBHf}9XeG;1`&;J5tHs2SX#j9Gj0HrOIB z-3cUME5S0N?x*J-R>(@?ZLvXe&VK(6?S9N-VmI=>wPsPfJgoL5M_~F$+<&U^o-*-V&lW9<{!+DifOh_R&^1+Df#7 zhw{D(St-^Fq$p_hd0dW}=mhD({uB;g@*5vLZ*hD3E-|2C$BdT*^wxCY(dre+?%STuFn%6#*8Z0Kp!Z*Br928mWk_wrqX=9fxFXdazr zg$5K|y@aOug;woV&OSdKyX6zBbc;pF_%rWN-Jiq}m1gkhmftW>&dF!oJPErNR&3D5 zTX!9I{5tl`<{D)0U7CiRk$LNs>~7E0{z9ZzN>0p!7f?)D_coDqf+8*-4AjdVYTLMk zFqDsfIW)yAJY%rD@6X3Y^1v+91?ai;n?1J@&wb+D_~Lqr{dS$jw>6iaT6HBU@7B6* zwhbcWFB~ZZNj)V)=W#<{;*hZW8-6w3{)5F55nsx@KiGF;DhNMs3B6 zbsnsn{7ALDo+fK52nS1sMBSaY$mNf%)5GW3U&L|$n353F*GFvMU_#sHAGA)aDvGLU z&dRFcq!o(o;OOLY28tKOv}8;kRs?j%755*V?3X;VE&rObeYZnGSzs$HyveP=&!{!~ z>O|iLZeu?#ScCGaFK(c4Tyxb>a{zYlu;+!ktep?jKmKlDuBrFnF1q*{6OqCwb-Y5W zsd|3?w6bwyo1O8&dvx5My+@S({hPWHhf_L-H5N5b823)MRoA=R8yIBN z@SN>~d&dj{m)gX0%Qd|G5@|#2$zvHGC_lC$1PyG3M%Phkv1v`E&1EeNSVD^Ram@J& zI~=c=kAF9U>r2VaCWOu$Yo{fiNopIK+1ODro!_^~bLJ*3Jm@x~=d}+l@_U!>&tE9* ztniR1OCsJE1xoyd^7tsX>D_r}soIqKBuL@dSqRjDc%XpikEZ<964^TQiIR9KMh+>; zSX?Wk%;5igKW%=swu=pCGQ2&xCN7EJx#g}`=&U?mP3jeLP3Eev{D^&_cJx)9tJkZX T8k8ku_2=v9=0mLC7m@ilud0$r literal 0 HcmV?d00001 diff --git a/docs/assays/metadata/merfish.md b/docs/assays/metadata/testing/merfish.md similarity index 100% rename from docs/assays/metadata/merfish.md rename to docs/assays/metadata/testing/merfish.md diff --git a/docs/assays/metadata/mplex.md b/docs/assays/metadata/testing/mplex.md similarity index 100% rename from docs/assays/metadata/mplex.md rename to docs/assays/metadata/testing/mplex.md diff --git a/docs/assays/metadata/olink.md b/docs/assays/metadata/testing/olink.md similarity index 100% rename from docs/assays/metadata/olink.md rename to docs/assays/metadata/testing/olink.md diff --git a/docs/assays/metadata/phenocycler.md b/docs/assays/metadata/testing/phenocycler.md similarity index 100% rename from docs/assays/metadata/phenocycler.md rename to docs/assays/metadata/testing/phenocycler.md diff --git a/docs/assays/metadata/pixel-seqv2.md b/docs/assays/metadata/testing/pixel-seqv2.md similarity index 100% rename from docs/assays/metadata/pixel-seqv2.md rename to docs/assays/metadata/testing/pixel-seqv2.md diff --git a/docs/assays/metadata/raman-imaging.md b/docs/assays/metadata/testing/raman-imaging.md similarity index 100% rename from docs/assays/metadata/raman-imaging.md rename to docs/assays/metadata/testing/raman-imaging.md diff --git a/docs/assays/metadata/secondharmonicgeneration.md b/docs/assays/metadata/testing/secondharmonicgeneration.md similarity index 100% rename from docs/assays/metadata/secondharmonicgeneration.md rename to docs/assays/metadata/testing/secondharmonicgeneration.md diff --git a/docs/assays/metadata/seq-scope.md b/docs/assays/metadata/testing/seq-scope.md similarity index 100% rename from docs/assays/metadata/seq-scope.md rename to docs/assays/metadata/testing/seq-scope.md diff --git a/docs/assays/metadata/seqfish.md b/docs/assays/metadata/testing/seqfish.md similarity index 100% rename from docs/assays/metadata/seqfish.md rename to docs/assays/metadata/testing/seqfish.md diff --git a/docs/assays/metadata/simple.md b/docs/assays/metadata/testing/simple.md similarity index 100% rename from docs/assays/metadata/simple.md rename to docs/assays/metadata/testing/simple.md diff --git a/docs/assays/metadata/sims.md b/docs/assays/metadata/testing/sims.md similarity index 100% rename from docs/assays/metadata/sims.md rename to docs/assays/metadata/testing/sims.md diff --git a/docs/assays/metadata/slide-seq.md b/docs/assays/metadata/testing/slide-seq.md similarity index 100% rename from docs/assays/metadata/slide-seq.md rename to docs/assays/metadata/testing/slide-seq.md diff --git a/docs/assays/metadata/snareseq2.md b/docs/assays/metadata/testing/snareseq2.md similarity index 100% rename from docs/assays/metadata/snareseq2.md rename to docs/assays/metadata/testing/snareseq2.md diff --git a/docs/assays/metadata/starmap.md b/docs/assays/metadata/testing/starmap.md similarity index 100% rename from docs/assays/metadata/starmap.md rename to docs/assays/metadata/testing/starmap.md diff --git a/docs/assays/metadata/thicksectionmultiphotonmxif.md b/docs/assays/metadata/testing/thicksectionmultiphotonmxif.md similarity index 100% rename from docs/assays/metadata/thicksectionmultiphotonmxif.md rename to docs/assays/metadata/testing/thicksectionmultiphotonmxif.md diff --git a/docs/assays/metadata/visium-hd.md b/docs/assays/metadata/testing/visium-hd.md similarity index 100% rename from docs/assays/metadata/visium-hd.md rename to docs/assays/metadata/testing/visium-hd.md diff --git a/docs/assays/metadata/visiumwithprobes.md b/docs/assays/metadata/testing/visiumwithprobes.md similarity index 100% rename from docs/assays/metadata/visiumwithprobes.md rename to docs/assays/metadata/testing/visiumwithprobes.md diff --git a/docs/assays/metadata/wgs.md b/docs/assays/metadata/testing/wgs.md similarity index 100% rename from docs/assays/metadata/wgs.md rename to docs/assays/metadata/testing/wgs.md diff --git a/scripts/newMeta2/source/reharmonize-legacy-metadata b/scripts/newMeta2/source/reharmonize-legacy-metadata deleted file mode 160000 index b184c78a..00000000 --- a/scripts/newMeta2/source/reharmonize-legacy-metadata +++ /dev/null @@ -1 +0,0 @@ -Subproject commit b184c78aa460ff60e961bf3aa4f5de7d0de53406 From b043c511dbf8901958df055f6dd4918b33b59291 Mon Sep 17 00:00:00 2001 From: Birdmachine Date: Fri, 17 Jul 2026 09:31:39 -0400 Subject: [PATCH 3/3] release harmonized metadata --- .../metadata/{testing => }/10X-Multiome.md | 0 docs/assays/metadata/10XMultiome.md | 24 - docs/assays/metadata/4i.md | 48 +- docs/assays/metadata/ATACseq.md | 351 ++++-------- .../{testing => }/Auto-fluorescence.md | 0 docs/assays/metadata/AutoFluorescence.md | 105 ---- docs/assays/metadata/CODEX.md | 182 +++--- docs/assays/metadata/COMET.md | 37 -- .../metadata/{testing => }/Cell-DIVE.md | 0 docs/assays/metadata/CosMx-Proteomics.md | 43 -- docs/assays/metadata/CosMx-Transcriptomics.md | 88 --- docs/assays/metadata/CyCIF.md | 34 -- docs/assays/metadata/CyTOF.md | 38 -- docs/assays/metadata/DESI.md | 112 ++-- docs/assays/metadata/DNA-Methylation.md | 28 - docs/assays/metadata/EnhancedSRS.md | 34 -- docs/assays/metadata/FACS.md | 39 -- docs/assays/metadata/GeoMx.md | 97 ---- docs/assays/metadata/HiFi.md | 47 -- docs/assays/metadata/Histology.md | 109 ++-- docs/assays/metadata/{testing => }/IMC-2D.md | 0 docs/assays/metadata/IMC.md | 234 -------- docs/assays/metadata/Illumina-Spatial.md | 41 -- docs/assays/metadata/LC-MS.md | 385 +++---------- .../metadata/{testing => }/Light-Sheet.md | 0 docs/assays/metadata/LightSheet.md | 146 ----- docs/assays/metadata/MALDI.md | 238 +++----- docs/assays/metadata/MERFISH.md | 47 -- docs/assays/metadata/MIBI.md | 190 +++---- docs/assays/metadata/MPLEx.md | 59 -- .../metadata/{testing => }/MUSIC-(CEDAR).md | 0 docs/assays/metadata/MUSIC.md | 128 +++-- docs/assays/metadata/Olink.md | 28 - docs/assays/metadata/PhenoCycler.md | 36 -- docs/assays/metadata/Pixel-seqV2.md | 37 -- .../{testing => }/RNAseq-(with-probes).md | 0 docs/assays/metadata/RNAseq.md | 520 ++++-------------- docs/assays/metadata/RNAseqWithProbes.md | 63 --- docs/assays/metadata/Raman-Imaging.md | 45 -- docs/assays/metadata/SIMS.md | 39 -- docs/assays/metadata/STARmap.md | 43 -- .../metadata/SecondHarmonicGeneration.md | 34 -- docs/assays/metadata/Seq-Scope.md | 39 -- docs/assays/metadata/Slide-seq.md | 94 ---- docs/assays/metadata/SnareSeq2.md | 19 - .../metadata/ThickSectionMultiphotonMxIF.md | 34 -- .../{testing => }/Visium-(no-probes).md | 0 docs/assays/metadata/Visium-HD.md | 33 -- docs/assays/metadata/VisiumNoProbes.md | 58 -- docs/assays/metadata/VisiumWithProbes.md | 30 - docs/assays/metadata/WGS.md | 86 --- docs/assays/metadata/{testing => }/comet.md | 0 .../{testing => }/cosmx-proteomics.md | 0 .../{testing => }/cosmx-transcriptomics.md | 0 docs/assays/metadata/{testing => }/cycif.md | 0 docs/assays/metadata/{testing => }/cytof.md | 0 .../metadata/{testing => }/dna-methylation.md | 0 .../metadata/{testing => }/enhancedsrs.md | 0 docs/assays/metadata/{testing => }/facs.md | 0 docs/assays/metadata/{testing => }/geomx.md | 0 docs/assays/metadata/{testing => }/hifi.md | 0 docs/assays/metadata/iCLAP.md | 34 -- docs/assays/metadata/{testing => }/iclap.md | 0 .../{testing => }/illumina-spatial.md | 0 docs/assays/metadata/{testing => }/imc.md | 0 docs/assays/metadata/index.md | 39 +- docs/assays/metadata/info.png | Bin 5196 -> 0 bytes docs/assays/metadata/info2.png | Bin 2024 -> 0 bytes docs/assays/metadata/link-circle.png | Bin 3955 -> 0 bytes docs/assays/metadata/link.png | Bin 3955 -> 0 bytes docs/assays/metadata/link2.png | Bin 1449 -> 0 bytes docs/assays/metadata/{testing => }/merfish.md | 0 docs/assays/metadata/{testing => }/mplex.md | 0 docs/assays/metadata/{testing => }/olink.md | 0 .../metadata/{testing => }/phenocycler.md | 0 .../metadata/{testing => }/pixel-seqv2.md | 0 .../metadata/{testing => }/raman-imaging.md | 0 .../{testing => }/secondharmonicgeneration.md | 0 .../metadata/{testing => }/seq-scope.md | 0 docs/assays/metadata/seqFISH.md | 90 --- docs/assays/metadata/{testing => }/seqfish.md | 0 docs/assays/metadata/{testing => }/simple.md | 0 docs/assays/metadata/{testing => }/sims.md | 0 .../metadata/{testing => }/slide-seq.md | 0 .../metadata/{testing => }/snareseq2.md | 0 docs/assays/metadata/{testing => }/starmap.md | 0 docs/assays/metadata/testing/.directory | 9 - docs/assays/metadata/testing/4i.md | 21 - docs/assays/metadata/testing/ATACseq.md | 94 ---- docs/assays/metadata/testing/CODEX.md | 62 --- docs/assays/metadata/testing/DESI.md | 69 --- docs/assays/metadata/testing/Histology.md | 68 --- docs/assays/metadata/testing/LC-MS.md | 89 --- docs/assays/metadata/testing/MALDI.md | 67 --- docs/assays/metadata/testing/MIBI.md | 86 --- docs/assays/metadata/testing/MUSIC.md | 70 --- docs/assays/metadata/testing/RNAseq.md | 95 ---- docs/assays/metadata/testing/index.md | 49 -- docs/assays/metadata/testing/info3.png | Bin 2038 -> 0 bytes .../thicksectionmultiphotonmxif.md | 0 .../metadata/{testing => }/visium-hd.md | 0 .../{testing => }/visiumwithprobes.md | 0 docs/assays/metadata/{testing => }/wgs.md | 0 103 files changed, 735 insertions(+), 4329 deletions(-) rename docs/assays/metadata/{testing => }/10X-Multiome.md (100%) delete mode 100644 docs/assays/metadata/10XMultiome.md rename docs/assays/metadata/{testing => }/Auto-fluorescence.md (100%) delete mode 100644 docs/assays/metadata/AutoFluorescence.md delete mode 100644 docs/assays/metadata/COMET.md rename docs/assays/metadata/{testing => }/Cell-DIVE.md (100%) delete mode 100644 docs/assays/metadata/CosMx-Proteomics.md delete mode 100644 docs/assays/metadata/CosMx-Transcriptomics.md delete mode 100644 docs/assays/metadata/CyCIF.md delete mode 100644 docs/assays/metadata/CyTOF.md delete mode 100644 docs/assays/metadata/DNA-Methylation.md delete mode 100644 docs/assays/metadata/EnhancedSRS.md delete mode 100644 docs/assays/metadata/FACS.md delete mode 100644 docs/assays/metadata/GeoMx.md delete mode 100644 docs/assays/metadata/HiFi.md rename docs/assays/metadata/{testing => }/IMC-2D.md (100%) delete mode 100644 docs/assays/metadata/IMC.md delete mode 100644 docs/assays/metadata/Illumina-Spatial.md rename docs/assays/metadata/{testing => }/Light-Sheet.md (100%) delete mode 100644 docs/assays/metadata/LightSheet.md delete mode 100644 docs/assays/metadata/MERFISH.md delete mode 100644 docs/assays/metadata/MPLEx.md rename docs/assays/metadata/{testing => }/MUSIC-(CEDAR).md (100%) delete mode 100644 docs/assays/metadata/Olink.md delete mode 100644 docs/assays/metadata/PhenoCycler.md delete mode 100644 docs/assays/metadata/Pixel-seqV2.md rename docs/assays/metadata/{testing => }/RNAseq-(with-probes).md (100%) delete mode 100644 docs/assays/metadata/RNAseqWithProbes.md delete mode 100644 docs/assays/metadata/Raman-Imaging.md delete mode 100644 docs/assays/metadata/SIMS.md delete mode 100644 docs/assays/metadata/STARmap.md delete mode 100644 docs/assays/metadata/SecondHarmonicGeneration.md delete mode 100644 docs/assays/metadata/Seq-Scope.md delete mode 100644 docs/assays/metadata/Slide-seq.md delete mode 100644 docs/assays/metadata/SnareSeq2.md delete mode 100644 docs/assays/metadata/ThickSectionMultiphotonMxIF.md rename docs/assays/metadata/{testing => }/Visium-(no-probes).md (100%) delete mode 100644 docs/assays/metadata/Visium-HD.md delete mode 100644 docs/assays/metadata/VisiumNoProbes.md delete mode 100644 docs/assays/metadata/VisiumWithProbes.md delete mode 100644 docs/assays/metadata/WGS.md rename docs/assays/metadata/{testing => }/comet.md (100%) rename docs/assays/metadata/{testing => }/cosmx-proteomics.md (100%) rename docs/assays/metadata/{testing => }/cosmx-transcriptomics.md (100%) rename docs/assays/metadata/{testing => }/cycif.md (100%) rename docs/assays/metadata/{testing => }/cytof.md (100%) rename docs/assays/metadata/{testing => }/dna-methylation.md (100%) rename docs/assays/metadata/{testing => }/enhancedsrs.md (100%) rename docs/assays/metadata/{testing => }/facs.md (100%) rename docs/assays/metadata/{testing => }/geomx.md (100%) rename docs/assays/metadata/{testing => }/hifi.md (100%) delete mode 100644 docs/assays/metadata/iCLAP.md rename docs/assays/metadata/{testing => }/iclap.md (100%) rename docs/assays/metadata/{testing => }/illumina-spatial.md (100%) rename docs/assays/metadata/{testing => }/imc.md (100%) delete mode 100644 docs/assays/metadata/info.png delete mode 100644 docs/assays/metadata/info2.png delete mode 100644 docs/assays/metadata/link-circle.png delete mode 100644 docs/assays/metadata/link.png delete mode 100644 docs/assays/metadata/link2.png rename docs/assays/metadata/{testing => }/merfish.md (100%) rename docs/assays/metadata/{testing => }/mplex.md (100%) rename docs/assays/metadata/{testing => }/olink.md (100%) rename docs/assays/metadata/{testing => }/phenocycler.md (100%) rename docs/assays/metadata/{testing => }/pixel-seqv2.md (100%) rename docs/assays/metadata/{testing => }/raman-imaging.md (100%) rename docs/assays/metadata/{testing => }/secondharmonicgeneration.md (100%) rename docs/assays/metadata/{testing => }/seq-scope.md (100%) delete mode 100644 docs/assays/metadata/seqFISH.md rename docs/assays/metadata/{testing => }/seqfish.md (100%) rename docs/assays/metadata/{testing => }/simple.md (100%) rename docs/assays/metadata/{testing => }/sims.md (100%) rename docs/assays/metadata/{testing => }/slide-seq.md (100%) rename docs/assays/metadata/{testing => }/snareseq2.md (100%) rename docs/assays/metadata/{testing => }/starmap.md (100%) delete mode 100644 docs/assays/metadata/testing/.directory delete mode 100644 docs/assays/metadata/testing/4i.md delete mode 100644 docs/assays/metadata/testing/ATACseq.md delete mode 100644 docs/assays/metadata/testing/CODEX.md delete mode 100644 docs/assays/metadata/testing/DESI.md delete mode 100644 docs/assays/metadata/testing/Histology.md delete mode 100644 docs/assays/metadata/testing/LC-MS.md delete mode 100644 docs/assays/metadata/testing/MALDI.md delete mode 100644 docs/assays/metadata/testing/MIBI.md delete mode 100644 docs/assays/metadata/testing/MUSIC.md delete mode 100644 docs/assays/metadata/testing/RNAseq.md delete mode 100644 docs/assays/metadata/testing/index.md delete mode 100644 docs/assays/metadata/testing/info3.png rename docs/assays/metadata/{testing => }/thicksectionmultiphotonmxif.md (100%) rename docs/assays/metadata/{testing => }/visium-hd.md (100%) rename docs/assays/metadata/{testing => }/visiumwithprobes.md (100%) rename docs/assays/metadata/{testing => }/wgs.md (100%) diff --git a/docs/assays/metadata/testing/10X-Multiome.md b/docs/assays/metadata/10X-Multiome.md similarity index 100% rename from docs/assays/metadata/testing/10X-Multiome.md rename to docs/assays/metadata/10X-Multiome.md diff --git a/docs/assays/metadata/10XMultiome.md b/docs/assays/metadata/10XMultiome.md deleted file mode 100644 index a1ab8b90..00000000 --- a/docs/assays/metadata/10XMultiome.md +++ /dev/null @@ -1,24 +0,0 @@ ---- -layout: page ---- -# 10X-Multiome - -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|----------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | True | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| True | -| number_of_pre-amplification_pcr_cycles | Numeric | The number of PCR cycles run after the Chromium Controller step and prior to separating the suspension and initiating library construction | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/4i.md b/docs/assays/metadata/4i.md index fe7bf3c1..ef03680f 100644 --- a/docs/assays/metadata/4i.md +++ b/docs/assays/metadata/4i.md @@ -1,37 +1,21 @@ ---- -layout: page --- -# 4i (Iterative Indirect Immunofluorescence Imaging) +layout: page-triary +--- + +# 4i Metadata Attributes -
Version 2 (current) +Fields that are collected for 4i data, available at ```dataset.metadata.``` +  -## Version 2 (current) +* indicates a required field -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| number_of_antibodies | Numeric | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | | True | -| number_of_biomarker_imaging_rounds | Numeric | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | | True | -| number_of_total_imaging_rounds | Numeric | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | | True | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| antibodies_path | Textfield | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | -| cell_boundary_marker_or_stain | Textfield | The name of the marker or stain used to identify all cell boundaries in the tissue. This name must exactly match the antibody-targeted molecule marker or non-antibody targeted molecule stain as found in the imaging data. For example, in the case of using the PhenoCycler, ensure the name corresponds to the value in the XPD output file. If multiple markers or stains are employed, list them in a comma-separated format. Example: Pan-Cytokeratin, E-Cadherin | | False | -| nuclear_marker_or_stain | Textfield | The nuclear marker or stain used, which can be an antibody-targeted molecule present in or around the cell nucleus. For protein targets, use the protein or gene symbol that identifies the antibody target, ensuring it matches the antibody target from the panel used or custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets, provide the stain name (e.g., DAPI) and, when applicable, include the associated staining kit and vendor. For the PhenoCycler, ensure the symbol matches the value found in the XPD output file. Example: DAPI | | False | -| number_of_channels | Numeric | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | True | +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| source_storage_duration_value * | | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | || time_since_acquisition_instrument_calibration_value | | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | +| contributors_path * | | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | || data_path * | | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | || number_of_antibodies * | | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | || number_of_biomarker_imaging_rounds * | | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | || number_of_total_imaging_rounds * | | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | || slide_id * | | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | || dataset_type * | | The specific type of dataset being produced. Example: RNAseq | ```Visium HD``` ```4i``` ```LC-MS``` ```Thick section Multiphoton MxIF``` ```Light Sheet``` ```ATACseq``` ```Resolve``` ```HiFi-Slide``` ```COMET``` ```MPLEx``` ```10X Multiome``` ```MALDI``` ```Histology``` ```Cell DIVE``` ```FACS``` ```MS Lipidomics``` ```Visium (no probes)``` ```MUSIC``` ```RNAseq``` ```GeoMx (NGS)``` ```GeoMx (nCounter)``` ```RNAseq (with probes)``` ```Singular Genomics G4X``` ```Molecular Cartography``` ```CosMx Transcriptomics``` ```MERFISH``` ```Pixel-seqV2``` ```2D Imaging Mass Cytometry``` ```Confocal``` ```seqFISH``` ```DART-FISH``` ```MIBI``` ```Olink``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```DESI``` ```Xenium``` ```CyCIF``` ```SNARE-seq2``` ```nanoSPLITS``` ```Stereo-seq``` ```Visium (with probes)``` ```SIMS``` ```Auto-fluorescence``` ```CyTOF``` ```CosMx Proteomics``` ```DBiT-seq``` ```PhenoCycler``` ```CODEX``` ```Second Harmonic Generation (SHG)``` ```Seq-Scope``` || analyte_class * | | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein``` ```Lipid + metabolite``` ```Collagen``` ```RNA``` ```Fluorochrome``` ```DNA``` ```Metabolite``` ```DNA + RNA``` ```Saturated lipid``` ```Lipid``` ```Peptide``` ```Protein``` ```Unsaturated lipid``` ```Endogenous fluorophore``` ```Chromatin``` ```Polysaccharide``` || acquisition_instrument_vendor * | | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics``` ```Cytek Biosciences``` ```Thermo Fisher Scientific``` ```Sciex``` ```Vizgen``` ```Leica Microsystems``` ```Akoya Biosciences``` ```Keyence``` ```Andor``` ```Standard BioTools (Fluidigm)``` ```Leica Biosystems``` ```Zeiss Microscopy``` ```Ionpath``` ```Motic``` ```In-House``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Element Biosciences``` ```Hamamatsu``` ```Bruker``` ```Illumina``` ```3DHISTECH``` ```Singular Genomics``` ```Huron Digital Pathology``` ```Resolve Biosciences``` ```NanoString``` ```Cytiva``` ```10x Genomics``` ```Microscopes International``` ```BGI Genomics``` || acquisition_instrument_model * | | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X``` ```NovaSeq X Plus``` ```Cytek Northern Lights``` ```Lightsheet 7``` ```Resolve Biosciences Molecular Cartography``` ```timsTOF HT``` ```timsTOF Pro 2``` ```timsTOF Pro``` ```timsTOF Ultra``` ```timsTOF Ultra 2``` ```timsTOF SCP``` ```Axio Scan.Z1``` ```MALDI timsTOF Flex Prototype``` ```CosMx Spatial Molecular Imager``` ```Unknown``` ```MERSCOPE Ultra``` ```Juno System``` ```timsTOF FleX``` ```Custom: Multiphoton``` ```CyTOF XT``` ```Helios``` ```EVOS M7000``` ```Aperio AT2``` ```Phenocycler-Fusion 2.0``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Observer 3``` ```NanoZoomer-SQ``` ```NanoZoomer S210``` ```NanoZoomer S60``` ```NanoZoomer S360``` ```DM6 B``` ```MoticEasyScan One``` ```In-House``` ```NextSeq 500``` ```BZ-X710``` ```QTRAP 5500``` ```NextSeq 550``` ```HiSeq 2500``` ```HiSeq 4000``` ```NovaSeq 6000``` ```Q Exactive HF``` ```Orbitrap Fusion Lumos Tribrid``` ```Q Exactive``` ```VS200 Slide Scanner``` ```Not applicable``` ```Orbitrap Eclipse Tribrid``` ```MIBIscope``` ```IN Cell Analyzer 2200``` ```timsTOF FleX MALDI-2``` || source_storage_duration_unit * | | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour``` ```month``` ```day``` ```minute``` ```year``` || time_since_acquisition_instrument_calibration_unit | | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month``` ```day``` ```year``` | +| metadata_schema_id * | | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | || preparation_protocol_doi * | | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | || is_targeted * | | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | || antibodies_path * | | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | || parent_sample_id * | | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | || non_global_files | | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | +| cell_boundary_marker_or_stain | | The name of the marker or stain used to identify all cell boundaries in the tissue. This name must exactly match the antibody-targeted molecule marker or non-antibody targeted molecule stain as found in the imaging data. For example, in the case of using the PhenoCycler, ensure the name corresponds to the value in the XPD output file. If multiple markers or stains are employed, list them in a comma-separated format. Example: Pan-Cytokeratin, E-Cadherin | | +| nuclear_marker_or_stain | | The nuclear marker or stain used, which can be an antibody-targeted molecule present in or around the cell nucleus. For protein targets, use the protein or gene symbol that identifies the antibody target, ensuring it matches the antibody target from the panel used or custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets, provide the stain name (e.g., DAPI) and, when applicable, include the associated staining kit and vendor. For the PhenoCycler, ensure the symbol matches the value found in the XPD output file. Example: DAPI | | +| number_of_channels * | | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | -
\ No newline at end of file diff --git a/docs/assays/metadata/ATACseq.md b/docs/assays/metadata/ATACseq.md index 3e030c10..f83141cb 100644 --- a/docs/assays/metadata/ATACseq.md +++ b/docs/assays/metadata/ATACseq.md @@ -1,257 +1,94 @@ ---- -layout: page ---- -# ATACseq - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - - -
Version 3 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | False | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | False | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina``` ```HiSeq 3000/4000 PE Cluster Kit PE-410-1001``` ```PN 1000283, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles)``` ```PN 20046811, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles)``` ```PN 20046812, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles)``` ```PN 20046813, Illumina``` ```NextSeq 2000 P3 Reagent Kit (300 Cycles)``` ```PN 20040561, Illumina``` ```NextSeq 2000 P3 Reagents Kit (100 Cycles)``` ```PN 20040559, Illumina``` ```NextSeq 500/550 Hi Output Kit 150 Cycles``` ```v2.5``` ```PN 20024907, Illumina``` ```NextSeq 500/550 Hi Output Kit 75 Cycles v2.5``` ```PN 20024906, Illumina``` ```NextSeq 500/550 Mid Output Kit 150 Cycles v2.5``` ```PN 20024904, Illumina``` ```NovaSeq 6000 S1 Reagent Kit (200 Cycles)``` ```PN 20012864, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028319, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028318, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles)``` ```PN 20028317, Illumina``` ```NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028316, Illumina``` ```NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles)``` ```PN 20028312, Illumina``` ```NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028313, Illumina``` ```NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles)``` ```PN 20028401, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (100 Cycle)``` ```PN 20104703, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (200 Cycle)``` ```PN 20104704, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (300 Cycle)``` ```PN 20104705, Illumina``` ```NovaSeq X Series 10B Reagent Kit (100 Cycle)``` ```PN 20085596, Illumina``` ```NovaSeq X Series 10B Reagent Kit (200 Cycle)``` ```PN 20085595, Illumina``` ```NovaSeq X Series 10B Reagent Kit (300 Cycle) ``` ```PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| transposition_method | Allowable Value | Modality of capturing accessible chromatin molecules. For example, this would be the type of kit that was used. | ```bulkATACseq``` ```sciATACseq``` ```Custom``` ```scATACseq``` | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | - -
- -
SNARE-seq2 / sciATACseq / snATACseq Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['SNARE-seq2', 'sciATACseq', 'snATACseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| is_technical_replicate | boolean | If TRUE, fastq files in dataset need to be merged. | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| sc_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol. | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. "OK" or "not OK", or with more specificity such as "debris", "clump", "low clump". | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment. | | True | -| transposition_input | Numeric | Number of cell/nuclei input to the assay. | | True | -| transposition_method | Allowable Value | Modality of capturing accessible chromatin molecules. | ['SNARE-Seq2-AC', 'bulkATACseq', 'snATACseq', 'sciATACseq'] | True | -| transposition_transposase_source | Allowable Value | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | ['10X snATAC', 'In-house', 'Nextera', '10X multiome'] | True | -| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". DOI for protocols.io referring to the protocol for this assay. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming. | | True | -| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). This field is not required for barcoding by single-cell combinatorial indexing. | | False | -| cell_barcode_offset | Textfield | Positions in the read at which the cell barcodes start. Cell barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. (Does not apply to sciATACseq, SNARE-seq and BulkATAC.) | | False | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs. Cell barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. (Does not apply to sciATACseq, SNARE-seq and BulkATAC.) | | False | -| library_pcr_cycles | Numeric | Number of PCR cycles to enrich for accessible chromatin fragments. | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library generation (figure in Descriptions section) | | True | -| library_final_yield | Numeric | Total ng of library after final pcr amplification step. | | True | -| library_final_yield_unit | Allowable Value | Units for library_final_yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
SNARE-seq2 / sciATACseq / snATACseq Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Textfield | The type of single cell entity derived from isolation protocol | | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Textfield | The method by which specific cell populations are sorted or enriched. | | False | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | boolean | Is the sequencing reaction run in repliucate, TRUE or FALSE | | True | -| cell_barcode_read | Textfield | Which read file contains the cell barcode | | True | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | True | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | True | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
bulkATACseq Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['bulkATACseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| bulk_transposition_input_number_nuclei | Textfield | A number (no comma separators) | | True | -| bulk_atac_cell_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | True | -| is_technical_replicate | boolean | Is this a sequencing replicate? | | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library_concentration_value | ['nM'] | False | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_creation_date | Datetime | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | False | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_pcr_cycles | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | True | -| library_preparation_kit | Textfield | Reagent kit used for library preparation | | True | -| sample_quality_metric | Textfield | This is a quality metric by visual inspection. This should answerthe question: Are the nuclei intact and are the nuclei free of significant amountsof debris? This can be captured at a high level, “OK” or “notOK”. | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | -| transposition_method | Textfield | Modality of capturing accessible chromatin molecules. The kit used, for example. | | True | -| transposition_transposase_source | Textfield | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- - - -
bulkATACseq 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['bulkATACseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | boolean | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| bulk_transposition_input_number_nuclei | Textfield | A number (no comma separators) | | True | -| bulk_atac_cell_isolation_protocols_io_doi | Textfield | Link to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | True | -| is_technical_replicate | boolean | Is this a sequencing replicate? | | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library_concentration_value | ['nM'] | False | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_creation_date | Datetime | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | False | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_pcr_cycles | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | True | -| library_preparation_kit | Textfield | Reagent kit used for library preparation | | True | -| sample_quality_metric | Textfield | This is a quality metric by visual inspection. This should answerthe question: Are the nuclei intact and are the nuclei free of significant amountsof debris? This can be captured at a high level, “OK” or “notOK”. | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | -| transposition_method | Textfield | Modality of capturing accessible chromatin molecules. The kit used, for example. | | True | -| transposition_transposase_source | Textfield | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
+--- +layout: page-triary +--- + +# ATACseq Metadata Attributes + +Fields that are collected for ATACseq data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| barcode_offset *| | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | +| barcode_read *| | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | +| barcode_size *| | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | +| umi_offset *| | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | +| umi_read *| | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | +| umi_size *| | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | +| assay_input_entity *| | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | +| number_of_input_cells_or_nuclei | | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | +| library_adapter_sequence *| | Adapter sequence to be used for adapter trimming | | +| library_average_fragment_size *| | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | +| library_input_amount_value | | The amount of cDNA, after amplification, that was used for library construction. | | +| library_input_amount_unit | | unit of library input amount value | ```ng``` ```ul``` | +| library_output_amount_value | | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | +| library_output_amount_unit | | Units of library final yield. | ```ng``` ```ul``` | +| library_concentration_value *| | The concentration value of the pooled library samples submitted for sequencing. | | +| library_concentration_unit *| | Unit of library_concentration_value | ```ng/ul``` ```nM``` | +| library_layout *| | State whether the library was generated for single-end or paired end sequencing. | ```paired-end``` ```single-end``` | +| number_of_pcr_cycles_for_indexing *| | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | +| library_preparation_kit *| | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```1 slides``` ```4 reactions; PN 1000338``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 reactions; PN 1000187``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | +| sample_indexing_kit *| | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001``` | +| sample_indexing_set *| | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | +| is_technical_replicate *| | Is this a sequencing replicate? | | +| expected_entity_capture_count | | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | +| sequencing_reagent_kit *| | Reagent kit used for sequencing | ```Custom``` ```Illumina``` ```HiSeq 3000/4000 PE Cluster Kit PE-410-1001``` ```PN 1000283, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles)``` ```PN 20046811, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles)``` ```PN 20046812, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles)``` ```PN 20046813, Illumina``` ```NextSeq 2000 P3 Reagent Kit (300 Cycles)``` ```PN 20040561, Illumina``` ```NextSeq 2000 P3 Reagents Kit (100 Cycles)``` ```PN 20040559, Illumina``` ```NextSeq 500/550 Hi Output Kit 150 Cycles``` ```v2.5``` ```PN 20024907, Illumina``` ```NextSeq 500/550 Hi Output Kit 75 Cycles v2.5``` ```PN 20024906, Illumina``` ```NextSeq 500/550 Mid Output Kit 150 Cycles v2.5``` ```PN 20024904, Illumina``` ```NovaSeq 6000 S1 Reagent Kit (200 Cycles)``` ```PN 20012864, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028319, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028318, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles)``` ```PN 20028317, Illumina``` ```NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028316, Illumina``` ```NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles)``` ```PN 20028312, Illumina``` ```NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028313, Illumina``` ```NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles)``` ```PN 20028401, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (100 Cycle)``` ```PN 20104703, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (200 Cycle)``` ```PN 20104704, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (300 Cycle)``` ```PN 20104705, Illumina``` ```NovaSeq X Series 10B Reagent Kit (100 Cycle)``` ```PN 20085596, Illumina``` ```NovaSeq X Series 10B Reagent Kit (200 Cycle)``` ```PN 20085595, Illumina``` ```NovaSeq X Series 10B Reagent Kit (300 Cycle)``` ```PN 20085594``` | +| sequencing_read_format *| | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | +| transposition_reagent_kit | | | | +| transposition_method *| | Modality of capturing accessible chromatin molecules. The kit used, for example. | ```bulkATACseq``` ```sciATACseq``` ```Custom``` ```scATACseq``` | +| sequencing_batch_id | | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| capture_batch_id | | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_instrument_kit | | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| library_construction_protocols_io_doi | | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | +| library_creation_date | | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | +| library_id | | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| sample_quality_metric | | This is a quality metric by visual inspection. This should answer the question: Are the nuclei intact and are the nuclei free of significant amounts of debris? This can be captured at a high level, “OK” or “not OK”. | | +| library_pcr_cycles | | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | +| bulk_atac_cell_isolation_protocols_io_doi | | Link to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | +| sc_isolation_enrichment | | The method by which specific cell populations are sorted or enriched. | ```none``` ```FACS``` | +| sc_isolation_protocols_io_doi | | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | +| sc_isolation_quality_metric | | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. "OK" or "not OK", or with more specificity such as "debris", "clump", "low clump". | | +| sc_isolation_tissue_dissociation | | The method by which tissues are dissociated into single cells in suspension. | | +| sc_isolation_cell_number | | Total number of cell/nuclei yielded post dissociation and enrichment. | | +| sequencing_phix_percent | | Percent PhiX loaded to the run | | +| sequencing_read_percent_q30 | | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | +| transposition_transposase_source | | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | ```10X snATAC``` ```In-house``` ```Nextera``` ```10X multiome``` | +| version | | Version of the schema to use when validating this metadata. | ```1``` | +| description | | Free-text description of this assay. | | diff --git a/docs/assays/metadata/testing/Auto-fluorescence.md b/docs/assays/metadata/Auto-fluorescence.md similarity index 100% rename from docs/assays/metadata/testing/Auto-fluorescence.md rename to docs/assays/metadata/Auto-fluorescence.md diff --git a/docs/assays/metadata/AutoFluorescence.md b/docs/assays/metadata/AutoFluorescence.md deleted file mode 100644 index 6abca683..00000000 --- a/docs/assays/metadata/AutoFluorescence.md +++ /dev/null @@ -1,105 +0,0 @@ ---- -layout: page ---- -# Auto-fluorescence - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 2 (Latest) - -## Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement | ```month``` ```day``` ```year``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
Version 1 - -## Version 1 - -| Attribute | Type | Description | AllowableValues | Required | -|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['AF'] | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that this is resolution within a given sample: z-pitch (resolution_z_value) is the increment distance between image slices, ie. the microscope stage is moved up or down in increments to capture images of several focal planes. | | True | -| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | -| number_of_channels | Numeric | Number of channels capturing the emission spectrum from natural fluorophores in the sample. | | True | -| overall_protocols_io_doi | Textfield | DOI for protocols.io referring to the overall protocol for the assay. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | AllowableValues | Required | -|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['AF'] | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that this is resolution within a given sample: z-pitch (resolution_z_value) is the increment distance between image slices, ie. the microscope stage is moved up or down in increments to capture images of several focal planes. | | True | -| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | -| number_of_channels | Numeric | Number of channels capturing the emission spectrum from natural fluorophores in the sample. | | True | -| overall_protocols_io_doi | Textfield | DOI for protocols.io referring to the overall protocol for the assay. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/CODEX.md b/docs/assays/metadata/CODEX.md index f2efa40f..4550574c 100644 --- a/docs/assays/metadata/CODEX.md +++ b/docs/assays/metadata/CODEX.md @@ -1,120 +1,62 @@ ---- -layout: page ---- -# CODEX - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 2 (Latest) - -## Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| number_of_biomarker_imaging_rounds | Numeric | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo application, 2. fluor application, 3. washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | True | -| number_of_total_imaging_rounds | Numeric | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| total_run_time_value | Numeric | How long the tissue was on the acquisition instrument. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| total_run_time_unit | Allowable Value | The units for the total run time unit field. | ```Hour``` ```Minute``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | True | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
Version 1 - -## Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['CODEX', 'CODEX2'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes', 'No'] | True | -| acquisition_instrument_vendor | Allowable Value | An acquisition_instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing molecular mass. | ['Keyence', 'Zeiss'] | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | ['BZ-X800', 'BZ-X710', 'Axio Observer Z1'] | True | -| resolution_x_value | Numeric | The width of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of width of a pixel.(nm) | ['mm', 'um', 'nm'] | False | -| resolution_y_value | Numeric | The height of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of height of a pixel. (nm) | ['mm', 'um', 'nm'] | False | -| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that thisis resolution within a given sample: z-pitch (resolution_z_value) is the incrementdistance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stageis moved up or down in increments of 1.5um to capture images of several focalplanes. The best one will be used & the rest discarded. The thickness of the sampleitself is sample metadata. | | False | -| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare the sample for theassay. | ['CODEX'] | True | -| preparation_instrument_model | Allowable Value | The model number/name of the instrument used to prepare the samplefor the assay | ['version 1 robot', 'prototype robot - Stanford/Nolan Lab'] | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| number_of_cycles | Numeric | Number of cycles of 1. oligo application, 2. fluor application, 3.washes | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagentsfor the assay. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['CODEX'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes', 'No'] | True | -| acquisition_instrument_vendor | Allowable Value | An acquisition_instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing molecular mass. | ['Keyence', 'Zeiss'] | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | ['BZ-X800', 'BZ-X710', 'Axio Observer Z1'] | True | -| resolution_x_value | Numeric | The width of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of width of a pixel.(nm) | ['mm', 'um', 'nm'] | False | -| resolution_y_value | Numeric | The height of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of height of a pixel. (nm) | ['mm', 'um', 'nm'] | False | -| resolution_z_value | Numeric | Optional if assay does not have multiple z-levels. Note that thisis resolution within a given sample: z-pitch (resolution_z_value) is the incrementdistance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stageis moved up or down in increments of 1.5um to capture images of several focalplanes. The best one will be used & the rest discarded. The thickness of the sampleitself is sample metadata. | | False | -| resolution_z_unit | Allowable Value | The unit of incremental distance between image slices. | ['mm', 'um', 'nm'] | False | -| | Textfield | | | | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare the sample for theassay. | ['CODEX'] | True | -| preparation_instrument_model | Allowable Value | The model number/name of the instrument used to prepare the samplefor the assay | ['version 1 robot', 'prototype robot - Stanford/Nolan Lab'] | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| number_of_cycles | Numeric | Number of cycles of 1. oligo application, 2. fluor application, 3.washes | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagentsfor the assay. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
+--- +layout: page-triary +--- + +# CODEX Metadata Attributes + +Fields that are collected for CODEX data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor *| | An acquisition_instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| antibodies_path *| | Relative path to file with antibody information for this dataset. | | +| preparation_instrument_vendor *| | The manufacturer of the instrument used to prepare the sample for the assay. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model *| | The model number/name of the instrument used to prepare the sample for the assay | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| total_run_time_value | | How long the tissue was on the acquisition instrument. | | +| total_run_time_unit | | The units for the total run time unit field. | ```Hour``` ```Minute``` | +| number_of_antibodies *| | Number of antibodies | | +| number_of_channels *| | Number of fluorescent channels imaged during each cycle. | | +| number_of_biomarker_imaging_rounds *| | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo application, 2. fluor application, 3. washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | +| number_of_total_imaging_rounds *| | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | +| slide_id | | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| resolution_z_unit | | The unit of incremental distance between image slices. | ```mm``` ```um``` ```nm``` | +| resolution_z_value | | Optional if assay does not have multiple z-levels. Note that this is resolution within a given sample: z-pitch (resolution_z_value) is the increment distance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stage is moved up or down in increments of 1.5um to capture images of several focal planes. The best one will be used & the rest discarded. The thickness of the sample itself is sample metadata. | | diff --git a/docs/assays/metadata/COMET.md b/docs/assays/metadata/COMET.md deleted file mode 100644 index 7a7348b1..00000000 --- a/docs/assays/metadata/COMET.md +++ /dev/null @@ -1,37 +0,0 @@ ---- -layout: page ---- -# COMET - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| number_of_antibodies | Numeric | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | | True | -| number_of_biomarker_imaging_rounds | Numeric | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | | True | -| number_of_total_imaging_rounds | Numeric | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | | True | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| antibodies_path | Textfield | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | -| cell_boundary_marker_or_stain | Textfield | The name of the marker or stain used to identify all cell boundaries in the tissue. This name must exactly match the antibody-targeted molecule marker or non-antibody targeted molecule stain as found in the imaging data. For example, in the case of using the PhenoCycler, ensure the name corresponds to the value in the XPD output file. If multiple markers or stains are employed, list them in a comma-separated format. Example: Pan-Cytokeratin, E-Cadherin | | False | -| nuclear_marker_or_stain | Textfield | The nuclear marker or stain used, which can be an antibody-targeted molecule present in or around the cell nucleus. For protein targets, use the protein or gene symbol that identifies the antibody target, ensuring it matches the antibody target from the panel used or custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets, provide the stain name (e.g., DAPI) and, when applicable, include the associated staining kit and vendor. For the PhenoCycler, ensure the symbol matches the value found in the XPD output file. Example: DAPI | | False | -| number_of_channels | Numeric | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/testing/Cell-DIVE.md b/docs/assays/metadata/Cell-DIVE.md similarity index 100% rename from docs/assays/metadata/testing/Cell-DIVE.md rename to docs/assays/metadata/Cell-DIVE.md diff --git a/docs/assays/metadata/CosMx-Proteomics.md b/docs/assays/metadata/CosMx-Proteomics.md deleted file mode 100644 index d50cdc02..00000000 --- a/docs/assays/metadata/CosMx-Proteomics.md +++ /dev/null @@ -1,43 +0,0 @@ ---- -layout: page ---- -# CosMx Proteomics - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | False | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | False | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | -| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | -| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | -| oligo_probe_panel | Assigned Value | The oligo probe panel used to target genes and/or proteins. If there is a core panel along with add-on modules, the core panel should be selected in this field. Any additional panels utilized should be documented in the "additional_panels_used.csv" file, which must be uploaded alongside the dataset. Example: 10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363 | ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-MsWTA-4```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (RNA, 1000 Plex); PN CMX-M-NEUP-R```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 16 rxns; PN 1000420```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 4 rxns x 4 BC; PN 1000475```, ```NanoString Technologies; CosMx Human 6K Discovery Panel (RNA, 6175 Plex); PN 121500041```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome, 4 rxns x 1 BC; PN 1000474```, ```10x Genomics; Xenium Human Multi-Tissue and Cancer Panel v1; PN 1000626```, ```NanoString Technologies; CosMx Human Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-H-USCP-1KP-R```, ```10x Genomics; Xenium Custom Gene Expression Panel (up to 50 genes); PN 1000464```, ```NanoString Technologies; CosMx Hs Univ Cell (RNA, 1000 Plex); PN 121500002```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365```, ```10x Genomics; Xenium Mouse Multi-Tissue Atlassing Panel; PN 1000627```, ```10x Genomics; Xenium Custom Gene Expression Panel (51-100 genes); PN 1000561```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 64 rxns; PN 1000456```, ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-HuWTA-4```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit v2.0 - Small; PN 1000667```, ```NanoString Technologies; CosMx Mouse Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-M-USCP-1KP-R```, ```NanoString Technologies; CosMx Human Immuno-Oncology Panel (Protein, 64 Plex); PN CMX-H-IOP-64P-P```, ```Custom```, ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466```, ```10x Genomics; Xenium Human Colon Gene Expression Panel; PN 1000642```, ```10x Genomics; Chromium Next GEM Single Cell Fixed RNA Mouse Transcriptome Probe Kit, 64 rxns; PN 1000492```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (Protein, 64 Plex); PN CMX-M-Neuro-64P-P```, ```NanoString Technologies; CosMx Hs WTX RNA Panel Kit, 2 slides: PN 121500047```, ```10x Genomics; Xenium Human Lung Gene Expression Panel; PN 1000601```, ```10x Genomics; Xenium Prime 5K Human Pan Tissue & Pathways Panel; PN 1000724``` | False | -| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | -| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | -| roi_label | Textfield | The label for the region of interest (ROI). For Resolve and CosMx, this corresponds to the field of view (FOV) label. In the case of Xenium, it refers to the ID of the region containing the analysis. For GeoMx, this information can be located in the "Initial Dataset" spreadsheet, which can be downloaded from within the Data Analysis Suite. Example: Decidua | | False | -| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | -| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/CosMx-Transcriptomics.md b/docs/assays/metadata/CosMx-Transcriptomics.md deleted file mode 100644 index 62695999..00000000 --- a/docs/assays/metadata/CosMx-Transcriptomics.md +++ /dev/null @@ -1,88 +0,0 @@ ---- -layout: page ---- -# CosMx Transcriptomics - -
Version 3 (current) - -## Version 3 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial v0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | False | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | False | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | -| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| proteinasek_concentration | Numeric | The concentration of the enzyme Proteinase K within a sample, measured in micrograms per milliliter (ug/ml). Example: 10 | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is incubated with Proteinase K. Example: 15 | | False | -| proteinasek_incubation_time_unit | Assigned Value | The unit of measurement for the proteinaseK incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | -| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | -| oligo_probe_panel | Assigned Value | The oligo probe panel used to target genes and/or proteins. If there is a core panel along with add-on modules, the core panel should be selected in this field. Any additional panels utilized should be documented in the "additional_panels_used.csv" file, which must be uploaded alongside the dataset. Example: 10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363 | ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-MsWTA-4```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 16 rxns x 16 BC; PN 1000547```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (RNA, 1000 Plex); PN CMX-M-NEUP-R```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 16 rxns; PN 1000420```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 4 rxns x 4 BC; PN 1000475```, ```NanoString Technologies; CosMx Human 6K Discovery Panel (RNA, 6175 Plex); PN 121500041```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome, 4 rxns x 1 BC; PN 1000474```, ```10x Genomics; Xenium Human Multi-Tissue and Cancer Panel v1; PN 1000626```, ```NanoString Technologies; CosMx Human Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-H-USCP-1KP-R```, ```10x Genomics; GEM-X Flex Human Transcriptome Probe Kit, 16 samples; PN 1000785```, ```10x Genomics; Xenium Custom Gene Expression Panel (up to 50 genes); PN 1000464```, ```NanoString Technologies; CosMx Hs Univ Cell (RNA, 1000 Plex); PN 121500002```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365```, ```10x Genomics; Xenium Mouse Multi-Tissue Atlassing Panel; PN 1000627```, ```10x Genomics; Xenium Custom Gene Expression Panel (51-100 genes); PN 1000561```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 64 rxns; PN 1000456```, ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-HuWTA-4```, ```NanoString Technologies; GeoMx Human IO Proteome Atlas, 4 slides; PN 121300160```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit v2.0 - Small; PN 1000667```, ```NanoString Technologies; CosMx Mouse Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-M-USCP-1KP-R```, ```NanoString Technologies; CosMx Human Immuno-Oncology Panel (Protein, 64 Plex); PN CMX-H-IOP-64P-P```, ```Custom```, ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466```, ```10x Genomics; Xenium Human Colon Gene Expression Panel; PN 1000642```, ```10x Genomics; Chromium Next GEM Single Cell Fixed RNA Mouse Transcriptome Probe Kit, 64 rxns; PN 1000492```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (Protein, 64 Plex); PN CMX-M-Neuro-64P-P```, ```NanoString Technologies; CosMx Hs WTX RNA Panel Kit, 2 slides: PN 121500047```, ```10x Genomics; Xenium Human Lung Gene Expression Panel; PN 1000601```, ```10x Genomics; Xenium Prime 5K Human Pan Tissue & Pathways Panel; PN 1000724``` | True | -| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | -| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | -| roi_label | Textfield | The label for the region of interest (ROI). For Resolve and CosMx, this corresponds to the field of view (FOV) label. In the case of Xenium, it refers to the ID of the region containing the analysis. For GeoMx, this information can be located in the "Initial Dataset" spreadsheet, which can be downloaded from within the Data Analysis Suite. Example: Decidua | | False | -| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | -| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | - -
- - -
Version 2 - -## Version 2 - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial v0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | False | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | False | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | -| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| proteinasek_concentration | Numeric | The concentration of the enzyme Proteinase K within a sample, measured in micrograms per milliliter (ug/ml). Example: 10 | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is incubated with Proteinase K. Example: 15 | | False | -| proteinasek_incubation_time_unit | Assigned Value | The unit of measurement for the proteinaseK incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | -| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | -| oligo_probe_panel | Assigned Value | The oligo probe panel used to target genes and/or proteins. If there is a core panel along with add-on modules, the core panel should be selected in this field. Any additional panels utilized should be documented in the "additional_panels_used.csv" file, which must be uploaded alongside the dataset. Example: 10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363 | ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-MsWTA-4```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 16 rxns x 16 BC; PN 1000547```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (RNA, 1000 Plex); PN CMX-M-NEUP-R```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 16 rxns; PN 1000420```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome 4 rxns x 4 BC; PN 1000475```, ```NanoString Technologies; CosMx Human 6K Discovery Panel (RNA, 6175 Plex); PN 121500041```, ```10x Genomics; Chromium Fixed RNA Kit, Human Transcriptome, 4 rxns x 1 BC; PN 1000474```, ```10x Genomics; Xenium Human Multi-Tissue and Cancer Panel v1; PN 1000626```, ```NanoString Technologies; CosMx Human Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-H-USCP-1KP-R```, ```10x Genomics; GEM-X Flex Human Transcriptome Probe Kit, 16 samples; PN 1000785```, ```10x Genomics; Xenium Custom Gene Expression Panel (up to 50 genes); PN 1000464```, ```NanoString Technologies; CosMx Hs Univ Cell (RNA, 1000 Plex); PN 121500002```, ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365```, ```10x Genomics; Xenium Mouse Multi-Tissue Atlassing Panel; PN 1000627```, ```10x Genomics; Xenium Custom Gene Expression Panel (51-100 genes); PN 1000561```, ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit, 64 rxns; PN 1000456```, ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas, 4 slides; PN GMX-RNA-NGS-HuWTA-4```, ```NanoString Technologies; GeoMx Human IO Proteome Atlas, 4 slides; PN 121300160```, ```10x Genomics; Visium Mouse Transcriptome Probe Kit v2.0 - Small; PN 1000667```, ```NanoString Technologies; CosMx Mouse Universal Cell Characterization Panel (RNA, 1000 Plex); PN CMX-M-USCP-1KP-R```, ```NanoString Technologies; CosMx Human Immuno-Oncology Panel (Protein, 64 Plex); PN CMX-H-IOP-64P-P```, ```Custom```, ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466```, ```10x Genomics; Xenium Human Colon Gene Expression Panel; PN 1000642```, ```10x Genomics; Chromium Next GEM Single Cell Fixed RNA Mouse Transcriptome Probe Kit, 64 rxns; PN 1000492```, ```NanoString Technologies; CosMx Mouse Neuroscience Panel (Protein, 64 Plex); PN CMX-M-Neuro-64P-P```, ```NanoString Technologies; CosMx Hs WTX RNA Panel Kit, 2 slides: PN 121500047```, ```10x Genomics; Xenium Human Lung Gene Expression Panel; PN 1000601```, ```10x Genomics; Xenium Prime 5K Human Pan Tissue & Pathways Panel; PN 1000724``` | True | -| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | -| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | -| roi_label | Textfield | The label for the region of interest (ROI). For Resolve and CosMx, this corresponds to the field of view (FOV) label. In the case of Xenium, it refers to the ID of the region containing the analysis. For GeoMx, this information can be located in the "Initial Dataset" spreadsheet, which can be downloaded from within the Data Analysis Suite. Example: Decidua | | False | -| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | -| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/CyCIF.md b/docs/assays/metadata/CyCIF.md deleted file mode 100644 index 2d293e15..00000000 --- a/docs/assays/metadata/CyCIF.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# CyCIF - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_biomarker_imaging_rounds | Numeric | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo and fluor application, 2. imaging, 3. removal of oligo and fluor along washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | True | -| number_of_total_imaging_rounds | Numeric | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/CyTOF.md b/docs/assays/metadata/CyTOF.md deleted file mode 100644 index f7f95499..00000000 --- a/docs/assays/metadata/CyTOF.md +++ /dev/null @@ -1,38 +0,0 @@ ---- -layout: page ---- -# CyTOF - -
Version 2 (Latest) - -## Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|----------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------|----------| -| lab_id | Textfield | An internal attribute labs can use to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This attribute will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Textfield | The specific type of dataset being produced. | | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Textfield | The time duration unit of measurement | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Textfield | The time unit of measurement | | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: [https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1](https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1). | | True | -| is_targeted | Assigned Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Custom``` ```None``` ```Sigma Aldrich; Cisplatin 25mg; PN P4394``` ```Standard BioTools; Cell-ID Cisplatin-198Pt 100 uL; PN 201198``` ```Standard BioTools; Cell-ID Intercalator-103Rh 2,000 um; PN 201103B``` ```Standard BioTools; Cell-ID Cisplatin-196Pt 100 uL; PN 201196``` ```Standard BioTools; Cell-ID Cisplatin 100 uL; PN 201064``` ```Standard BioTools; Cell-ID Intercalator-103Rh 500 um; PN 201103A``` ```Standard BioTools; Cell-ID Cisplatin-194Pt 100 uL; PN 201194``` ```Standard BioTools; Cell-ID Cisplatin-195Pt 100 uL; PN 201195``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| number_of_mass_channels | Numeric | The number of mass channels that measure the expression of markers in single cells. | | False | -| is_erythrocyte_lysis_performed | Assigned Value | Process in which red blood cells (RBCs) are broken down in the sample prior to analysis, thereby allowing researchers to focus primarily on white blood cells (WBCs). | ```Yes``` ```No``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | False | -| antibody_reagent_kit | Textfield | The kit containing the set of antibodies pre-conjugated with different heavy metal isotopes used to simultaneously detect and quantify multiple protein markers on individual cells by attaching these metal-labeled antibodies to specific cellular targets, essentially acting as the key component for labeling cells with the various markers needed for analysis on the CyTOF machine. | | False | -| viability_reagent_kit | Textfield | The kit used to differentiate between live and dead cells within a sample by selectively staining dead cells with a dye that can be detected by the instrument, allowing researchers to exclude dead cell data from their analysis and ensure accurate results when studying cell populations. | | False | -| is_cell_activation_performed | Assigned Value | Process by which ligand is binded to its receptors on a cell, which enhances the cell's ability to respond to various stimuli. | ```Yes``` ```No```e | False | -| activation_stimulus | Textfield | Specific type of stimulus used to provoke cell activation. Examples would include PMA/ionomycin or CD28in/brefeldin A. This field is required if "is_cells_activation performed" is Yes. | | False | -| is_fcr_blocking_applied | Assigned Value | Process by which a reagent has been added to the staining procedure to block the binding of antibodies to Fc receptors (FcRs) on cells, preventing non-specific binding and ensuring that only the intended target antigen is detected by the antibodies; essentially, it helps to minimize false positive signals by preventing antibodies from attaching to the cell via their Fc region instead of the antigen-specific binding site. | ```Yes``` ```No``` | False | -| is_heparin_used | Assigned Value | Indicates whether heparin was used ("Yes") or not ("No") during staining to prevent non-specific binding of metal-labeled antibodies to eosinophils to reduce background noise. |```Yes``` ```No``` | False | -| loaded_cell_concentration_value | Numeric | The number of cells present within a given volume of liquid for the experiment immediately prior to the experiment, essentially indicating how densely packed the cells are in a solution. | | False | -| loaded_cell_concentration_unit | Textfield | Unit of measure for cell concentration, e.g. cells per milliliter (cells/mL). | | False | -| instrument_calibration_bead_kit | Textfield | A set of beads of known mass intensity used to adjust the settings of a flow cytometer to ensure accurate measurements. | | False | -| calibration_kit_lot_number | Textfield | Manufacturer's lot number for the calibration bead kit used for the experiment. | | False | diff --git a/docs/assays/metadata/DESI.md b/docs/assays/metadata/DESI.md index 7ead4ea6..c651b8f3 100644 --- a/docs/assays/metadata/DESI.md +++ b/docs/assays/metadata/DESI.md @@ -1,43 +1,69 @@ ---- -layout: page ---- -# DESI - -
Version 2 (Latest) - -## Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | -| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | True | -| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | True | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | True | -| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | -| matrix_deposition_method | Allowable Value | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| preparation_matrix | Allowable Value | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| ion_mobility | Allowable Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```cIMS``` ```DTIMS``` ```FAIMS``` ```SLIM``` ```TIMS``` ```TWIMS``` | False | -| desorption_solvent | Allowable Value | Solvent composition for conducting nanospray desorption electrospray ionization (nanoDESI) or desorption electrospray ionization (DESI). | ```Acetonitrile:Dimethylformamide (ACN:DMF)``` ```Acetonitrile:Water (ACN:H2O)``` ```Ethanol:Dimethylformamide (EtOH:DMF)``` ```Ethanol:Water (EtOH:H2O)``` ```Methanol:Ethanol (MeOH:EtOH)``` ```Methanol:Water (MeOH:H2O)``` | True | -| desorption_solvent_flow_rate_value | Numeric | The rate of flow of the solvent into a spray. | | True | -| desorption_solvent_flow_rate_unit | Allowable Value | Units of the rate of solvent flow. | ```nL/min``` ```uL/min``` | True | -| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | -| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
+--- +layout: page-triary +--- + +# DESI Metadata Attributes + +Fields that are collected for DESI data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | +| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | +| ms_ionization_technique *| | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20``` | +| ms_scan_mode *| | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | +| mass_analysis_polarity *| | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | +| mass_to_charge_range_low_value | | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_to_charge_range_high_value | | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_resolving_power *| | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | +| mass_to_charge_resolving_power | | The peak (m/z) used to calculate the resolving power. | | +| ion_mobility | | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```TIMS``` ```SLIM``` ```FAIMS``` ```DTIMS``` ```cIMS``` ```TWIMS``` | +| matrix_deposition_method | | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_matrix | | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | +| desorption_solvent *| | Solvent composition for conducting nanospray desorption electrospray ionization (nanoDESI) or desorption electrospray ionization (DESI). | ```Acetonitrile:Dimethylformamide (ACN:DMF)``` ```Acetonitrile:Water (ACN:H2O)``` ```Ethanol:Dimethylformamide (EtOH:DMF)``` ```Ethanol:Water (EtOH:H2O)``` ```Methanol:Ethanol (MeOH:EtOH)``` ```Methanol:Water (MeOH:H2O)``` | +| desorption_solvent_flow_rate_value *| | The rate of flow of the solvent into a spray. | | +| desorption_solvent_flow_rate_unit *| | Units of the rate of solvent flow. | ```nL/min``` ```uL/min``` | +| analysis_protocol_doi *| | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| description | | Free-text description of this assay. | | +| overall_protocols_io_doi | | DOI for protocols.io referring to the overall protocol for the assay. | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/DNA-Methylation.md b/docs/assays/metadata/DNA-Methylation.md deleted file mode 100644 index f87faab0..00000000 --- a/docs/assays/metadata/DNA-Methylation.md +++ /dev/null @@ -1,28 +0,0 @@ ---- -layout: page ---- -# DNA Methylation - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial ver0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```DNA Methylation```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: d70bfe24-e82a-46cb-9369-28ae03660d97 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/EnhancedSRS.md b/docs/assays/metadata/EnhancedSRS.md deleted file mode 100644 index 7a906777..00000000 --- a/docs/assays/metadata/EnhancedSRS.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# Enhanced-SRS - -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
diff --git a/docs/assays/metadata/FACS.md b/docs/assays/metadata/FACS.md deleted file mode 100644 index 8c2d8d98..00000000 --- a/docs/assays/metadata/FACS.md +++ /dev/null @@ -1,39 +0,0 @@ ---- -layout: page ---- -# FACS - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| is_erythrocyte_lysis_performed | Radio | Process in which red blood cells (RBCs) are broken down in the sample prior to analysis, thereby allowing researchers to focus primarily on white blood cells (WBCs). | ```Yes,No``` | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | False | -| antibody_reagent_kit | Assigned Value | The kit containing the set of antibodies pre-conjugated with different heavy metal isotopes used to simultaneously detect and quantify multiple protein markers on individual cells by attaching these metal-labeled antibodies to specific cellular targets, essentially acting as the key component for labeling cells with the various markers needed for analysis on the CyTOF machine. | ```Standard BioTools; Maxpar Nuclear Antigen Staining Kit; PN 201603```, ```Standard BioTools; Maxpar Phosphoprotein Staining Kit; PN 201604```, ```Standard BioTools; Maxpar Cell Surface Staining Kit; PN 201601```, ```Standard BioTools; Maxpar Cytoplasmic/Secreted Antigen Staining Kit; PN 201602```, ```Custom``` | True | -| viability_reagent_kit | Assigned Value | The kit used to differentiate between live and dead cells within a sample by selectively staining dead cells with a dye that can be detected by the instrument, allowing researchers to exclude dead cell data from their analysis and ensure accurate results when studying cell populations. | ```Sigma Aldrich; Cisplatin 25mg; PN P4394```, ```Standard BioTools; Cell-ID Cisplatin-198Pt 100 uL; PN 201198```, ```None```, ```Standard BioTools; Cell-ID Intercalator-103Rh 2,000 um; PN 201103B```, ```Standard BioTools; Cell-ID Cisplatin-196Pt 100 uL; PN 201196```, ```Standard BioTools; Cell-ID Cisplatin 100 uL; PN 201064```, ```Standard BioTools; Cell-ID Intercalator-103Rh 500 um; PN 201103A```, ```Standard BioTools; Cell-ID Cisplatin-194Pt 100 uL; PN 201194```, ```Standard BioTools; Cell-ID Cisplatin-195Pt 100 uL; PN 201195```, ```Custom``` | True | -| is_cell_activation_performed | Radio | Process by which ligand is binded to its receptors on a cell, which enhances the cell's ability to respond to various stimuli. | ```Yes,No``` | True | -| activation_stimulus | Textfield | Specific type of stimulus used to provoke cell activation. Examples would include PMA/ionomycin or CD28in/brefeldin A. This field is required if "is_cells_activation performed" is Yes. | | False | -| is_fcr_blocking_applied | Radio | Process by which a reagent has been added to the staining procedure to block the binding of antibodies to Fc receptors (FcRs) on cells, preventing non-specific binding and ensuring that only the intended target antigen is detected by the antibodies; essentially, it helps to minimize false positive signals by preventing antibodies from attaching to the cell via their Fc region instead of the antigen-specific binding site. | ```Yes,No``` | True | -| is_heparin_used | Radio | Indicates whether heparin was used ("Yes") or not ("No") during staining to prevent non-specific binding of metal-labeled antibodies to eosinophils to reduce background noise. | ```Yes,No``` | True | -| loaded_cell_concentration_value | Numeric | The number of cells present within a given volume of liquid for the experiment immediately prior to the experiment, essentially indicating how densely packed the cells are in a solution. | | False | -| loaded_cell_concentration_unit | Assigned Value | Unit of measure for cell concentration, e.g. cells per milliliter (cells/mL). | ```cells/mL``` | False | -| instrument_calibration_bead_kit | Assigned Value | A set of beads of known mass intensity used to adjust the settings of a flow cytometer to ensure accurate measurements. | ```Standard BioTools; EQ Six Element Calibration Beads 100 mL; PN 201245```, ```Standard BioTools; EQ Four Element Calibration Beads 100 mL; PN 201078```, ```None```, ```Standard BioTools; CyTOF Calibration Beads; PN 201073```, ```Custom``` | True | -| calibration_kit_lot_number | Textfield | Manufacturer's lot number for the calibration bead kit used for the experiment. | | True | - -
diff --git a/docs/assays/metadata/GeoMx.md b/docs/assays/metadata/GeoMx.md deleted file mode 100644 index 0747a560..00000000 --- a/docs/assays/metadata/GeoMx.md +++ /dev/null @@ -1,97 +0,0 @@ ---- -layout: page ---- -# GeoMx - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
NGS Version 2 (current) - -## NGS Version 2 (current) - -| attribute | type | description | value | required | -|-----------------------------------------------------|----------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------|------------| -| dataset_type | Textfield | The specific type of dataset being produced. | | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Textfield | The time duration unit of measurement | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Textfield | The time unit of measurement | | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ['Yes', 'No'] | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Textfield | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | -| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | True | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | True | -| target_retrieval_incubation_time_unit | Textfield | The units for target retrieval incubation time value. | | True | -| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | False | -| proteinasek_incubation_time_unit | Textfield | The units for proteinaseK incubation time value. | | False | -| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | True | -| is_roi_segmentation_performed | Allowable Value | Was the image segmented. For GeoMx this refers to whether segmentation was used to split ROIs (regions of interest) into AOIs (areas of interest). | ['Yes', 'No'] | True | -| roi_segmentation_strategy | Textfield | The method of segmentation that was applied in a GeoMx assay. If an overlay was used the overlay image needs to be included in the dataset upload. | | False | -| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | -| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | -| targeted_entity_label | Textfield | State what cell type(s) or functional tissue unit was targeted in this ROI/AOI. | | True | -| targeted_entity_id | Textfield | The ontology ID for the targeted entity. | | False | -| segment_id | Textfield | This is the ID for the area of interest (AOI) in a GeoMx dataset. From "Initial Dataset" spreadsheet (download from within Data Analysis Suite), e.g. 9a828e39-43d8-4051-9bcc-581a520a85d4. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ['Yes', 'No'] | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | True | - -
- -
nCounter Version 2 (current) - -## nCounter Version 2 (current) - -| attribute | type | description | value | required | -|-----------------------------------------------------|----------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------|------------| -| dataset_type | Textfield | The specific type of dataset being produced. | | True | -| analyte_class | Textfield | Analytes are the target molecules being measured with the assay. | | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Textfield | The time duration unit of measurement | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Textfield | The time unit of measurement | | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ['Yes', 'No'] | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Textfield | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | -| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | True | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | True | -| target_retrieval_incubation_time_unit | Textfield | The units for target retrieval incubation time value. | | True | -| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | False | -| proteinasek_incubation_time_unit | Textfield | The units for proteinaseK incubation time value. | | False | -| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | True | -| is_roi_segmentation_performed | Allowable Value | Was the image segmented. For GeoMx this refers to whether segmentation was used to split ROIs (regions of interest) into AOIs (areas of interest). | ['Yes', 'No'] | True | -| roi_segmentation_strategy | Textfield | The method of segmentation that was applied in a GeoMx assay. If an overlay was used the overlay image needs to be included in the dataset upload. | | False | -| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | -| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | -| targeted_entity_label | Textfield | State what cell type(s) or functional tissue unit was targeted in this ROI/AOI. | | True | -| targeted_entity_id | Textfield | The ontology ID for the targeted entity. | | False | -| segment_id | Textfield | This is the ID for the area of interest (AOI) in a GeoMx dataset. From "Initial Dataset" spreadsheet (download from within Data Analysis Suite), e.g. 9a828e39-43d8-4051-9bcc-581a520a85d4. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ['Yes', 'No'] | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| hybcode_pack_lot_number | Textfield | Enter the lot number noted within the LabWorksheet.txt file (and used in downstream nCounter processing). | | True | -| probe_hybridization_time_value | Numeric | How many hours were the oligo-conjugated RNA or oligo-conjugated antibody probes hybridized with the sample? | | True | -| probe_hybridization_time_unit | Textfield | The units for probe hybridization time value. | | True | -| oligo_probe_panel | Textfield | This is the probe panel used to target genes and/or proteins. In cases where there is a core panel and add-on modules, the core panel should be selected here. If additional panels are used, then they must be included in the "additional_panels_used.csv" file that's uploaded with the dataset. | | True | -| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ['Yes', 'No'] | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | True | - -
diff --git a/docs/assays/metadata/HiFi.md b/docs/assays/metadata/HiFi.md deleted file mode 100644 index ca89eb5c..00000000 --- a/docs/assays/metadata/HiFi.md +++ /dev/null @@ -1,47 +0,0 @@ ---- -layout: page ---- -# HiFi - -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | True | -| spot_size_value | Numeric | FModified progressive staining, Not applicable, Progressive staining, Regressive stainingor assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Allowable Value | The unit for spot size value. | ```um^2``` ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Allowable Value | Units corresponding to inter-spot distance | ```um``` | True | -| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be ```A1, B1, C1, D1```. For HiFi this would be the lane on the flowcell. | ```A1``` ```B1``` ```C1``` ```D1``` ```Lane 1``` ```Lane 2``` ```Lane 3``` ```Lane 4``` ```Lane 5``` ```Lane 6``` ```Lane 7``` ```Lane 8``` | False | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | True | -| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | True | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | True | -| target_retrieval_incubation_time_unit | Allowable Value | The units for target retrieval incubation time value. | ```minute``` | True | -| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | True | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | True | -| proteinasek_incubation_time_unit | Allowable Value | The units for proteinaseK incubation time value. | ```minute``` | True | -| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | -| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | False | - -
diff --git a/docs/assays/metadata/Histology.md b/docs/assays/metadata/Histology.md index 7170d0bb..b02db691 100644 --- a/docs/assays/metadata/Histology.md +++ b/docs/assays/metadata/Histology.md @@ -1,41 +1,68 @@ ---- -layout: page ---- -# Histology - -
Version 2 (Latest) - -## Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| is_batch_staining_done | Allowable Value | Are the slides stained using a linear batch method or individually? | ```Yes``` ```No``` | True | -| is_staining_automated | Allowable Value | Is the slide staining automated with an instrument? | ```Yes``` ```No``` | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| stain_name | Allowable Value | The name of the chemical stains (dyes) applied to histology samples to highlight important features of the tissue as well as to enhance the tissue contrast. | ```AB-PAS``` ```H&E``` ```H-DAB``` ```LFB``` ```PAS``` ```Trichrome ```| True | -| stain_technique | Allowable Value | There are typically three types of stains: progressive, modified progressive, and regressive. Progressive staining occurs when the hematoxylin is added to the tissue without being followed by a differentiator to remove excess dye. With regressive and modified progressive staining, a differentiator is used. | ```Modified progressive staining``` ```Not applicable``` ```Progressive staining``` ```Regressive staining``` | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | - -
+--- +layout: page-triary +--- + +# Histology Metadata Attributes + +Fields that are collected for Histology data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | +| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | +| is_image_preprocessing_required | | Indicates whether image preprocessing is necessary based on the type of acquisition instrument used, such as a microscope or slide scanner. This may involve steps like fusing image tiles to assemble the complete image. Example: Yes | | +| stain_name *| | The name of the chemical stains (dyes) applied to histology samples to highlight important features of the tissue as well as to enhance the tissue contrast. | ```AB-PAS``` ```H&E``` ```H-DAB``` ```LFB``` ```PAS``` ```Trichrome``` | +| stain_technique | | There are typically three types of stains: progressive, modified progressive, and regressive. Progressive staining occurs when the hematoxylin is added to the tissue without being followed by a differentiator to remove excess dye. With regressive and modified progressive staining, a differentiator is used. | ```Modified progressive staining``` ```Not applicable``` ```Progressive staining``` ```Regressive staining``` | +| is_batch_staining_done *| | Are the slides stained using a linear batch method or individually? | | +| is_staining_automated *| | Is the slide staining automated with an instrument? | | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| slide_id | | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| tile_configuration | | The configuration of tiles used for stitching in the assay process. If no tile configuration is applicable, enter "Not applicable". Example: Row-by-row | ```Column-by-column``` ```Not applicable``` ```Snake-by-columns``` ```Row-by-row``` ```Snake-by-rows``` | +| scan_direction | | The direction of imaging, which is necessary for the stitching process. Example: Left-and-down | ```Left-and-down``` ```Right-and-down``` ```Not applicable``` ```Right-and-up``` ```Left-and-up``` | +| tiled_image_columns | | The number of columns used in the stitching process of a tiled image, often referred to as the grid size in the x-dimension. Example: 5 | | +| tiled_image_count | | The total number of raw tiled images captured, which are intended to be stitched together. Example: 75 | | +| intended_tile_overlap_percentage | | The intended percentage of overlap between tiled images. This value serves as the set point, although slight variations may occur during image acquisition due to stage registration. Example: 5 | | +| non_global_files | | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| overall_protocols_io_doi | | DOI for protocols.io referring to the overall protocol for the assay. | | +| principal_investigator | | | | +| pi_email | | Email address for the principal investigator. | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| resolution_z_unit | | The unit of incremental distance between image slices. | ```mm``` ```um``` ```nm``` | +| resolution_z_value | | Optional if assay does not have multiple z-levels. Note that thisis resolution within a given sample: z-pitch (resolution_z_value) is the incrementdistance between image slices (for Akoya, z-pitch=1.5um) ie. the microscope stageis moved up or down in increments of 1.5um to capture images of several focalplanes. The best one will be used & the rest discarded. The thickness of the sampleitself is sample metadata. | | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/testing/IMC-2D.md b/docs/assays/metadata/IMC-2D.md similarity index 100% rename from docs/assays/metadata/testing/IMC-2D.md rename to docs/assays/metadata/IMC-2D.md diff --git a/docs/assays/metadata/IMC.md b/docs/assays/metadata/IMC.md deleted file mode 100644 index d4383200..00000000 --- a/docs/assays/metadata/IMC.md +++ /dev/null @@ -1,234 +0,0 @@ ---- -layout: page ---- -# IMC-2D - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
2D IMC Version 2 (Latest) - -## 2D IMC Version 2 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| total_run_time_value | Numeric | How long the tissue was on the acquisition instrument. | | True | -| total_run_time_unit | Allowable Value | The units for the total run time unit field. | ```Hour``` ```Minute``` | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes. | | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ```Hz``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - - -
- -
2D IMC Version 1 - -## 2D IMC Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Imaging Mass Cytometry'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| number_of_channels | Numeric | Number of mass channels measured | | True | -| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | -| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | -| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| dual_count_start | Numeric | Threshold for dual counting. | | True | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
2D IMC Version 0 - -## 2D IMC Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Imaging Mass Cytometry'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| number_of_channels | Numeric | Number of mass channels measured | | True | -| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | -| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | -| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| dual_count_start | Numeric | Threshold for dual counting. | | True | -| end_datetime | Datetime | Time stamp indicating end of ablation for ROI | | True | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| start_datetime | Datetime | Time stamp indicating start of ablation for ROI | | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
3D IMC Version 1 (no longer accepting data) - -## 3D IMC Version 1 (no longer accepting data) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['3D Imaging Mass Cytometry'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| number_of_channels | Numeric | Number of mass channels measured | | True | -| number_of_sections | Numeric | Number of sections | | True | -| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | -| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | -| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
3D IMC Version 0 - -## 3D IMC Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['3D Imaging Mass Cytometry'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| number_of_channels | Numeric | Number of mass channels measured | | True | -| number_of_sections | Numeric | Number of sections | | True | -| ablation_distance_between_shots_x_value | Numeric | x resolution. Distance between laser ablation shots in the X-dimension. | | True | -| ablation_distance_between_shots_x_units | Allowable Value | Units of x resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_distance_between_shots_y_value | Numeric | y resolution. Distance between laser ablation shots in the Y-dimension. | | True | -| ablation_distance_between_shots_y_units | Allowable Value | Units of y resolution distance between laser ablation shots. | ['um', 'nm'] | True | -| ablation_frequency_value | Numeric | Frequency value of laser ablation (in Hz) | | True | -| ablation_frequency_unit | Allowable Value | Frequency unit of laser ablation | ['Hz'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/Illumina-Spatial.md b/docs/assays/metadata/Illumina-Spatial.md deleted file mode 100644 index 498c6115..00000000 --- a/docs/assays/metadata/Illumina-Spatial.md +++ /dev/null @@ -1,41 +0,0 @@ ---- -layout: page ---- -# Illumina Spatial ver0 - -
Version 0 (current) - -## Version 0 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```Illumina Spatial v0```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```MACSima```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | True | -| capture_area_id | Radio | The capture area on the slide that was used during the process. For example, in the case for Visium, this would correspond to areas such as [A1, B1, C1, D1], while for HiFi, it would refer to the lane on the flowcell. Example: A1 | ```A1```, ```B1```, ```C1```, ```D1```, ```Lane 1```, ```Lane 2```, ```Lane 3```, ```Lane 4```, ```Lane 5```, ```Lane 6```, ```Lane 7```, ```Lane 8``` | False | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| preparation_instrument_vendor | Assigned Value | The company that manufactures the instrument used to prepare the sample (e.g., for staining or other processing steps) prior to the assay. If the instrument was custom-built or developed internally, enter "In-House". If no sample preparation occurred, enter "Not applicable". Example: 10X Genomics | ```Thermo Fisher Scientific```, ```SunChrom```, ```Akoya Biosciences```, ```Leica Biosystems```, ```Ionpath```, ```Roche Diagnostics```, ```In-House```, ```Not applicable```, ```Hamamatsu```, ```HTX Technologies```, ```10x Genomics``` | False | -| preparation_instrument_model | Assigned Value | The specific model of the instrument used for sample preparation, such as staining. Manufacturers may offer multiple models with varying features or sensitivities, which can influence how the sample is processed and how the resulting data is interpreted. If no sample preparation occurred, enter "Not applicable". Example: Chromium X | ```AutoStainer XL```, ```ST5020 Multistainer```, ```Visium CytAssist```, ```SunCollect Sprayer```, ```Chromium X```, ```Chromium iX```, ```EVOS M7000```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```Discovery Ultra```, ```Sublimator```, ```Not applicable```, ```TM-Sprayer```, ```M5 Sprayer```, ```M3+ Sprayer```, ```Chromium Controller```, ```Chromium Connect```, ```Custom``` | False | -| capture_area_width_value | Numeric | The width of RNA capture area. Example: 10 | | True | -| capture_area_width_unit | Assigned Value | The unit of measurement for the capture area width value. If the width value is not specified, this field may be left blank. Example: mm | ```mm``` | True | -| capture_area_height_value | Numeric | The height of RNA capture area. Example: 10 | | True | -| capture_area_height_unit | Assigned Value | The unit of measurement for the capture area height value. If the height value is not specified, this field may be left blank. Example: mm | ```mm``` | True | -| spatial_discreatization_method | Assigned Value | The segmentation method used to divide the capture are into smaller, defined regions for analysis. Example: Cell segmentation | ```Square binning```, ```Cell segmentation```, ```Hexagonal binning``` | True | -| bin_size | Textfield | The size (in µm) of each discrete spatial unit ("bin") used to partition the capture area in bin-based spatial discretization. Example: 100 | | False | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Miltenyi Biotec```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```LSM 710 Confocal Microscope```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```MACSima System```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive HF-X``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/LC-MS.md b/docs/assays/metadata/LC-MS.md index 75cdce06..a98bf37b 100644 --- a/docs/assays/metadata/LC-MS.md +++ b/docs/assays/metadata/LC-MS.md @@ -1,296 +1,89 @@ ---- -layout: page ---- -# LC-MS - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 4 (Latest) - -## Version 4 (Latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | -| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | False | -| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | -| ion_mobility | Allowable Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```cIMS``` ```DTIMS``` ```FAIMS``` ```SLIM``` ```TIMS``` ```TWIMS``` | False | -| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | -| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | -| label_name | Textfield | If the samples were labeled (e.g. TMT), provide the name/ID of the label on this sample. Leave blank if not applicable. | | False | -| lc_instrument_vendor | Allowable Value | The manufacturer of the instrument used for liquid chromatography. | ```Agilent Technologies``` ```Bruker``` ```Evosep``` ```In-House``` ```Sciex``` ```Thermo Fisher Scientific``` ```Waters``` | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for liquid chromatography. | | False | -| lc_column_model | Textfield | The model number/name of the liquid chromatography column. If it is a custom self-packed, pulled tip capillary is used enter “Pulled tip capilary”. | | False | -| lc_resin | Textfield | Details of the resin used for liquid chromatography, including vendor, particle size, pore size. | | False | -| lc_column_length_value | Numeric | Liquid chromatography column length. | | False | -| lc_column_length_unit | Allowable Value | Units for liquid chromatography column length (typically cm). | ```um``` ```mm``` ```cm``` | False | -| lc_temperature_value | Numeric | Liquid chromatography temperature. | | False | -| lc_inner_diameter_value | Numeric | Liquid chromatography column inner diameter. | | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_gradient_value | Numeric | Liquid chromatography gradient. | | False | -| lc_gradient_unit | Allowable Value | Unit for liquid chromatography gradient | ```Minute``` | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A. | | False | -| lc_mobile_phase_b | Textfield | | | False | -| spatial_sampling_technique | Allowable Value | | ```LCM``` ```LESA``` ```microLESA``` ```microPOTS``` ```nanoPOTS``` ```nanoSPLITS``` | False | -| spatial_sampling_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targeted in the spatial profiling experiment. Leave blank if data are generated in imaging mode without a specific target structure. | | False | -| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependent acquisition), DIA (Data-independent acquisition), SRM (multiple reaction monitoring), or PRM (parallel reaction monitoring). | ```DDA``` ```PRM``` ```DIA``` ```SRM``` | False | -| lc_column_vendor | Allowable Value | The manufacturer of the liquid chromatography column unless self-packed, pulled tip capillary is used. | ```Bruker``` ```Evosep``` ```In-House``` ```IonOpticks``` ```Thermo Fisher Scientific``` ```Waters``` | False | -| lc_temperature_unit | Allowable Value | | ```Celsius``` | False | -| lc_inner_diameter_unit | Allowable Value | | ```um``` ```mm``` ```cm``` | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ```mL/min``` ```nL/min``` | False | -| spatial_sampling_type | Allowable Value | Specifies whether or not the analysis was performed in a spatially targeted manner. Spatial profiling experiments target specific tissue foci but do not necessarily generate images. Spatial imaging expriments collect data from a regular array (pixels) that can be visualized as heat maps of ion intensity at each location (molecular images). Leave blank if data are derived from bulk analysis. | ```Imaging``` ```Profiling``` | False | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
Version 3 - -## Version 3 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['3'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | -| assay_type | Allowable Value | Bottom-up refers to analyzing proteins in a sample by digesting themto peptides. Top-down refers to analyzing whole proteins without digestion. LC-MSand MS are for lipids/metabolites. LC-MS Bottom-Up and MS Bottom-Up are for peptides.LC-MS Top-Down and MS Top-Down are for proteins. | ['LC-MS', 'MS', 'LC-MS Bottom-Up', 'MS Bottom-Up', 'LC-MS Top-Down', 'MS Top-Down'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids', 'peptides', 'phosphopeptides', 'glycans'] | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| dms | Allowable Value | Was differential mobility spectrometry used in this assay? | ['Yes','No'] | True | -| ms_source | Allowable Value | The ion source type used for surface sampling. | ['ESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| mass_resolving_power | Numeric | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | False | -| mz_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | -| ion_mobility | Allowable Value | Specifies whether or not ion mobility spectrometry was performed andwhich technology was used. Technologies for measuring ion mobility: TravelingWave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS),High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube IonMobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ['TIMS', 'TWIMS', 'FAIMS', 'DTIMS', 'SLIMS'] | False | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | -| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | -| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | -| label_name | Textfield | If the samples were labeled (e.g. TMT), provide the name/ID of thelabel on this sample. | | False | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | -| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | -| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | -| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | -| lc_length_value | Numeric | LC column length | | False | -| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | -| lc_temp_value | Numeric | LC temperature | | False | -| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | -| lc_id_value | Numeric | LC column inner diameter (microns) | | False | -| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | -| lc_gradient | Textfield | LC gradient | | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | -| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | -| spatial_type | Allowable Value | Specifies whether or not the analysis was performed in a spatialy targetedmanner and the technique used for spatial sampling. For example, Laser-capturemicrodissection (LCM), Liquid Extraction Surface Analysis (LESA), NanodropletProcessing in One pot for Trace Samples (nanoPOTS). | ['LCM', 'LESA', 'nanoPOTS', 'microLESA'] | False | -| spatial_sampling_type | Allowable Value | Specifies whether or not the analysis was performed in a spatiallytargeted manner. Spatial profiling experiments target specific tissue foci butdo not necessarily generate images. Spatial imaging expriments collect data froma regular array (pixels) that can be visualized as heat maps of ion intensityat each location (molecular images). Leave blank if data are derived from bulkanalysis. | ['profiling', 'imaging'] | False | -| spatial_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targetedin the spatial profiling experiment. Leave blank if data are generated in imagingmode without a specific target structure. | | False | -| resolution_x_value | Numeric | The width of a pixel. | | False | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | False | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 2 - -## Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | -| assay_type | Allowable Value | Bottom-up refers to analyzing proteins in a sample by digesting themto peptides. Top-down refers to analyzing whole proteins without digestion. LC-MSand MS are for lipids/metabolites. LC-MS Bottom-Up and MS Bottom-Up are for peptides.LC-MS Top-Down and MS Top-Down are for proteins. | ['LC-MS', 'MS', 'LC-MS Bottom-Up', 'MS Bottom-Up', 'LC-MS Top-Down', 'MS Top-Down'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids', 'peptides', 'phosphopeptides', 'glycans'] | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| ms_source | Allowable Value | The ion source type used for surface sampling. | ['ESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| mass_resolving_power | Numeric | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | False | -| mz_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | -| ion_mobility | Allowable Value | Specifies whether or not ion mobility spectrometry was performed andwhich technology was used. Technologies for measuring ion mobility: TravelingWave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS),High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube IonMobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ['TIMS', 'TWIMS', 'FAIMS', 'DTIMS', 'SLIMS'] | False | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | -| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | -| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | -| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | -| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | -| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | -| lc_length_value | Numeric | LC column length | | False | -| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | -| lc_temp_value | Numeric | LC temperature | | False | -| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | -| lc_id_value | Numeric | LC column inner diameter (microns) | | False | -| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | -| lc_gradient | Textfield | LC gradient | | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | -| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | -| spatial_type | Allowable Value | Specifies whether or not the analysis was performed in a spatialy targetedmanner and the technique used for spatial sampling. For example, Laser-capturemicrodissection (LCM), Liquid Extraction Surface Analysis (LESA), NanodropletProcessing in One pot for Trace Samples (nanoPOTS). | ['LCM', 'LESA', 'nanoPOTS', 'microLESA'] | False | -| spatial_sampling_type | Allowable Value | Specifies whether or not the analysis was performed in a spatiallytargeted manner. Spatial profiling experiments target specific tissue foci butdo not necessarily generate images. Spatial imaging expriments collect data froma regular array (pixels) that can be visualized as heat maps of ion intensityat each location (molecular images). Leave blank if data are derived from bulkanalysis. | ['profiling', 'imaging'] | False | -| spatial_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targetedin the spatial profiling experiment. Leave blank if data are generated in imagingmode without a specific target structure. | | False | -| resolution_x_value | Numeric | The width of a pixel. | | False | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | False | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 1 - -## Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['LC-MS (metabolomics)', 'LC-MS/MS (label-free proteomics)', 'MS (shotgun lipidomics)'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| ms_source | Textfield | The ion source type used for surface sampling (MALDI, MALDI-2, DESI,or SIMS) or LC-MS/MS data acquisition (nESI) | | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | -| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | -| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | -| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | -| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | -| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | -| lc_length_value | Numeric | LC column length | | False | -| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | -| lc_temp_value | Numeric | LC temperature | | False | -| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | -| lc_id_value | Numeric | LC column inner diameter (microns) | | False | -| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | -| lc_gradient | Textfield | LC gradient | | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | -| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | -| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-----------------------------------------------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['mass_spectrometry'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['LC-MS (metabolomics)', 'LC-MS/MS (label-free proteomics)', 'MS (shotgun lipidomics)'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| ms_source | Textfield | The ion source type used for surface sampling (MALDI, MALDI-2, DESI,or SIMS) or LC-MS/MS data acquisition (nESI) | | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| data_collection_mode | Allowable Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependentacquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring),or PRM (parallel reaction monitoring). | ['DDA', 'DIA', 'MRM', 'PRM'] | True | -| ms_scan_mode | Textfield | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 forTMT) | | True | -| labeling | Textfield | Indicates whether samples were labeled prior to MS analysis (e.g.,TMT) | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissuesections for the assay. | | True | -| lc_instrument_vendor | Textfield | The manufacturer of the instrument used for LC | | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for LC | | False | -| lc_column_vendor | Textfield | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulledtip capilary is used | | False | -| lc_column_model | Textfield | The model number/name of the LC Column - IF custom self-packed, pulledtip calillary is used enter "Pulled tip capilary" | | False | -| lc_resin | Textfield | Details of the resin used for lc, including vendor, particle size,pore size | | False | -| lc_length_value | Numeric | LC column length | | False | -| lc_length_unit | Allowable Value | units for LC column length (typically cm) | ['um', 'mm', 'cm'] | False | -| lc_temp_value | Numeric | LC temperature | | False | -| lc_temp_unit | Allowable Value | units for LC temperature | ['C'] | False | -| lc_id_value | Numeric | LC column inner diameter (microns) | | False | -| lc_id_unit | Allowable Value | units of LC column inner diameter (typically microns) | ['um', 'mm', 'cm'] | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_flow_rate_unit | Allowable Value | Units of flow rate. | ['nL/min', 'mL/min'] | False | -| lc_gradient | Textfield | LC gradient | | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A | | False | -| lc_mobile_phase_b | Textfield | Composition of mobile phase B | | False | -| processing_search | Textfield | Software for analyzing and searching LC-MS/MS omics data | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process for this assay. | | False | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
\ No newline at end of file +--- +layout: page-triary +--- + +# LC-MS Metadata Attributes + +Fields that are collected for LC-MS data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| ms_ionization_technique *| | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20``` | +| ms_scan_mode *| | Indicates whether experiment is MS, MS/MS, or other (possibly MS3 for TMT) | ```MS1``` ```MS2``` ```MS3``` | +| mass_analysis_polarity *| | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | +| mass_to_charge_range_low_value | | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_to_charge_range_high_value | | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_resolving_power | | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | +| mass_to_charge_resolving_power | | The peak (m/z) used to calculate the resolving power. | | +| ion_mobility | | Specifies whether or not ion mobility spectrometry was performed and which technology was used. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ```TIMS``` ```SLIM``` ```FAIMS``` ```DTIMS``` ```cIMS``` ```TWIMS``` | +| data_collection_mode *| | Mode of data collection in tandem MS assays. Either DDA (Data-dependent acquisition), DIA (Data-independent acquisition), MRM (multiple reaction monitoring), or PRM (parallel reaction monitoring). | ```DDA``` ```PRM``` ```DIA``` ```SRM``` | +| label_name | | If the samples were labeled (e.g. TMT), provide the name/ID of the label on this sample. | | +| lc_instrument_vendor | | The manufacturer of the instrument used for LC | ```Thermo Fisher Scientific``` ```Sciex``` ```In-House``` ```Agilent Technologies``` ```Waters``` ```Bruker``` ```Evosep``` | +| lc_instrument_model | | The model number/name of the instrument used for LC | | +| lc_column_vendor | | OPTIONAL: The manufacturer of the LC Column unless self-packed, pulled tip capilary is used | ```Thermo Fisher Scientific``` ```In-House``` ```Waters``` ```Bruker``` ```Evosep``` ```IonOpticks``` | +| lc_column_model | | The model number/name of the LC Column - IF custom self-packed, pulled tip calillary is used enter "Pulled tip capilary" | | +| lc_resin | | Details of the resin used for lc, including vendor, particle size, pore size | | +| lc_column_length_value | | Liquid chromatography column length. | | +| lc_column_length_unit | | Units for liquid chromatography column length (typically cm). | ```um``` ```mm``` ```cm``` | +| lc_temperature_value | | Liquid chromatography temperature. | | +| lc_temperature_unit | | | ```celsius``` | +| lc_inner_diameter_value | | Liquid chromatography column inner diameter. | | +| lc_inner_diameter_unit | | | ```um``` ```mm``` ```cm``` | +| lc_flow_rate_value | | Value of flow rate. | | +| lc_flow_rate_unit | | Units of flow rate. | ```nL/min``` ```mL/min``` | +| lc_gradient_value | | Liquid chromatography gradient. | | +| lc_gradient_unit | | Unit for liquid chromatography gradient | ```minute``` | +| lc_mobile_phase_a | | Composition of mobile phase A | | +| lc_mobile_phase_b | | Composition of mobile phase B | | +| spatial_sampling_technique | | | ```nanoSPLITS``` ```nanoPOTS``` ```LESA``` ```microPOTS``` ```LCM``` ```microLESA``` | +| spatial_sampling_target | | Specifies the cell-type or functional tissue unit (FTU) that is targeted in the spatial profiling experiment. Leave blank if data are generated in imaging mode without a specific target structure. | | +| spatial_sampling_type | | Specifies whether or not the analysis was performed in a spatially targeted manner. Spatial profiling experiments target specific tissue foci but do not necessarily generate images. Spatial imaging expriments collect data from a regular array (pixels) that can be visualized as heat maps of ion intensity at each location (molecular images). Leave blank if data are derived from bulk analysis. | ```Imaging``` ```Profiling``` | +| analysis_protocol_doi *| | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | +| acquisition_protocol_doi | | | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| description | | Free-text description of this assay. | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| protocols_io_doi | | DOI for protocols.io referring to the protocol for this assay. | | +| overall_protocols_io_doi | | DOI for protocols.io for the overall process for this assay. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| processing_search | | Software for analyzing and searching LC-MS/MS omics data | | +| labeling | | Indicates whether samples were labeled prior to MS analysis (e.g., TMT) | | +| dms | | Was differential mobility spectrometry used in this assay? | | +| resolution_x_unit | | The unit of measurement of the width of a pixel. | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. | | +| resolution_y_unit | | The unit of measurement of the height of a pixel. | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel | | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/testing/Light-Sheet.md b/docs/assays/metadata/Light-Sheet.md similarity index 100% rename from docs/assays/metadata/testing/Light-Sheet.md rename to docs/assays/metadata/Light-Sheet.md diff --git a/docs/assays/metadata/LightSheet.md b/docs/assays/metadata/LightSheet.md deleted file mode 100644 index 4eb8950d..00000000 --- a/docs/assays/metadata/LightSheet.md +++ /dev/null @@ -1,146 +0,0 @@ ---- -layout: page ---- -# Light-Sheet - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 3 (latest) - -## Version 3 (latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
Version 2 - -## Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|-------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Light Sheet'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| range_z_value | Numeric | The total range of the z axis. | | True | -| range_z_unit | Allowable Value | The unit of range_z_value. | ['nm', 'um'] | False | -| step_z_value | Numeric | The number of optical sections in z axis range. | | True | -| increment_z_value | Numeric | The distance between sequential optical sections. | | True | -| increment_z_unit | Allowable Value | The units of increment z value. | ['nm', 'um'] | False | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 1 - -## Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Light Sheet'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| resolution_z_value | Numeric | The distance at which two objects along the detection z-axis can bedistinguished (resolved as 2 objects). | | True | -| resolution_z_unit | Allowable Value | The unit of distance at which two objects along the detection z-axiscan be distinguished (resolved as 2 objects). | ['mm', 'um', 'nm'] | False | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped foldergenerated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year,MM is the month with leading 0s, and DD is the day with leading 0s, hh is thehour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories:generation of images of microscopic entities, identification & quantitation ofmolecules by mass spectrometry, imaging mass spectrometry, and determination ofnucleotide sequence. | ['imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Light Sheet'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted fordetection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detectionhardware and signal processing software. Assays generate signals such as lightof various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions(models) of that instrument with different features or sensitivities. Differencesin features or sensitivities may be relevant to processing or interpretation ofthe data. | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| resolution_z_value | Numeric | The distance at which two objects along the detection z-axis can bedistinguished (resolved as 2 objects). | | True | -| resolution_z_unit | Allowable Value | The unit of distance at which two objects along the detection z-axiscan be distinguished (resolved as 2 objects). | ['mm', 'um', 'nm'] | False | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstreamprocessing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/MALDI.md b/docs/assays/metadata/MALDI.md index ccf79e7c..ae760058 100644 --- a/docs/assays/metadata/MALDI.md +++ b/docs/assays/metadata/MALDI.md @@ -1,171 +1,67 @@ ---- -layout: page ---- -# MALDI - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Maldi Version 2 (latest) - -## Maldi Version 2 (latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | True | -| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | -| ion_mobility | Allowable Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```cIMS``` ```DTIMS``` ```FAIMS``` ```SLIM``` ```TIMS``` ```TWIMS``` | False | -| matrix_deposition_method | Allowable Value | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| preparation_matrix | Allowable Value | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | -| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
- -
IMS Version 2 - -## IMS Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['MALDI-IMS', 'SIMS-IMS', 'NanoDESI', 'DESI'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids', 'peptides', 'phosphopeptides', 'glycans'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| ms_source | Allowable Value | The ion source type used for surface sampling (MALDI, MALDI-2, DESI, nanoDESI or SIMS). | ['MALDI', 'MALDI-2', 'LDI', 'LA', 'SIMS-C60', 'SIMS-H2O', 'DESI', 'nanoDESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| mass_resolving_power | Numeric | The MS1 resolving power defined as m/∆m where ∆m is the FWHM for a given peak with a specified m/z (m). (unitless) | | True | -| mz_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | -| ion_mobility | Allowable Value | Specifies whether or not ion mobility spectrometry was performed and which technology was used. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS, Structures for Lossless Ion Manipulations (SLIM). | ['TIMS', 'TWIMS', 'FAIMS', 'DTIMS', 'SLIMS'] | False | -| ms_scan_mode | Allowable Value | Scan mode refers to the number of steps in the separation of fragments. | ['MS', 'MS/MS', 'MS3'] | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| preparation_type | Textfield | Common methods of depositing matrix for MALDI imaging include robotic spotting, electrospray deposition, and spray-coating with an airbrush. | | False | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | False | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | False | -| preparation_maldi_matrix | Textfield | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the laser. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | | False | -| desi_solvent | Textfield | Solvent composition for conducting nanospray desorption electrospray ionization (nanoDESI) or desorption electrospray ionization (DESI). | | False | -| desi_solvent_flow_rate | Numeric | The rate of flow of the solvent into a spray. | | False | -| desi_solvent_flow_rate_unit | Allowable Value | Units of the rate of solvent flow. | ['uL/minute'] | False | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| processing_protocols_io_doi | Textfield | DOI for analysis protocols.io for this assay. | | False | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
IMS Version 1 - -## IMS Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['MALDI-IMS'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| ms_source | Allowable Value | The ion source type used for surface sampling (MALDI, MALDI-2, DESI, or SIMS) or LC-MS/MS data acquisition (nESI) | ['MALDI', 'MALDI-2', 'DESI', 'SIMS', 'nESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| preparation_type | Textfield | Common methods of depositing matrix for MALDI imaging include robotic spotting, electrospray deposition, and spray-coating with an airbrush. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| preparation_maldi_matrix | Textfield | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the laser. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
IMS Version 0 - -## IMS Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['MALDI-IMS'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein', 'metabolites', 'lipids'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| ms_source | Allowable Value | The ion source type used for surface sampling (MALDI, MALDI-2, DESI, or SIMS) or LC-MS/MS data acquisition (nESI) | ['MALDI', 'MALDI-2', 'DESI', 'SIMS', 'nESI'] | True | -| polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes) | ['negative ion mode', 'positive ion mode', 'negative and positive ion mode'] | True | -| mz_range_low_value | Numeric | The low value of the scanned mass range for MS1. (unitless) | | True | -| mz_range_high_value | Numeric | The high value of the scanned mass range for MS1. (unitless) | | True | -| resolution_x_value | Numeric | The width of a pixel. | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of the width of a pixel. | ['nm', 'um'] | False | -| resolution_y_value | Numeric | The height of a pixel | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of the height of a pixel. | ['nm', 'um'] | False | -| preparation_type | Textfield | Common methods of depositing matrix for MALDI imaging include robotic spotting, electrospray deposition, and spray-coating with an airbrush. | | True | -| preparation_instrument_vendor | Textfield | The manufacturer of the instrument used to prepare the sample for the assay. | | True | -| preparation_instrument_model | Textfield | The model number/name of the instrument used to prepare the sample for the assay | | True | -| preparation_maldi_matrix | Textfield | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the laser. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| overall_protocols_io_doi | Textfield | DOI for protocols.io for the overall process. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
+--- +layout: page-triary +--- + +# MALDI Metadata Attributes + +Fields that are collected for MALDI data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | +| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | +| ms_ionization_technique *| | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20``` | +| ms_scan_mode *| | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | +| mass_analysis_polarity *| | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | +| mass_to_charge_range_low_value | | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_to_charge_range_high_value | | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | +| mass_resolving_power *| | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | +| mass_to_charge_resolving_power | | The peak (m/z) used to calculate the resolving power. | | +| ion_mobility | | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```TIMS``` ```SLIM``` ```FAIMS``` ```DTIMS``` ```cIMS``` ```TWIMS``` | +| matrix_deposition_method *| | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | +| preparation_instrument_vendor *| | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model *| | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_matrix *| | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | +| analysis_protocol_doi *| | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| description | | Free-text description of this assay. | | +| section_prep_protocols_io_doi | | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | +| overall_protocols_io_doi | | DOI for protocols.io referring to the overall protocol for the assay. | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/MERFISH.md b/docs/assays/metadata/MERFISH.md deleted file mode 100644 index bb457b7d..00000000 --- a/docs/assays/metadata/MERFISH.md +++ /dev/null @@ -1,47 +0,0 @@ ---- -layout: page ---- -# MERFISH -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | False | -| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | False | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| target_retrieval_incubation_temperature | Numeric | Will normally be 100 degrees Celsius for RNA assays, and 80 degrees Celsius for protein assays. | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. | | False | -| target_retrieval_incubation_time_unit | Allowable Value | The units for target retrieval incubation time value. | ```minute``` | False | -| proteinasek_concentration | Numeric | The amount or concentration of the enzyme Proteinase K within a sample (in ug/ml). | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is exposed to Proteinase K. | | False | -| proteinasek_incubation_time_unit | Allowable Value | The units for proteinaseK incubation time value. | ```minute``` | False | -| probe_hybridization_time_value | Numeric | How long was the oligo-conjugated RNA or oligo-conjugated antibody probes hybridized with the sample? | | False | -| probe_hybridization_time_unit | Allowable Value | The units for probe hybridization time value. | ```Hour``` ```Minute``` | False | -| oligo_probe_panel | Allowable Value | This is the probe panel used to target genes and/or proteins. In cases where there is a core panel and add-on modules, the core panel should be selected here. If additional panels are used, then they must be included in the "additional_panels_used.csv" file that's uploaded with the dataset. | ```10x Genomics; Chromium Fixed RNA Kit``` ```Human Transcriptome``` ```4 rxns x 1 BC; PN 1000474``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```16 rxns; PN 1000420``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```64 rxns; PN 1000456``` ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363``` ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365``` ```Custom``` ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-HuWTA-4``` ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-MsWTA-4``` | True | -| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ```Yes``` ```No``` | True | -| number_of_panel_targets | Numeric | How many genes, RNA isoforms or RNA regions are targeted by probes. | | True | -| roi_label | Textfield | A label for the region of interest (ROI). For Xenium, Resolve and CosMx, this is the field of view (FOV) label. For GeoMx this can be found in the "Initial Dataset" spreadsheet (download from within Data Analysis Suite). | | False | -| anatomical_structure_label | Textfield | The overarching anatomical structure. | | False | -| anatomical_structure_id | Textfield | The ontology ID for the parent structure. Typically this would be an UBERON ID. | | False | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | -| number_of_additional_stains | Numeric | This would be minimally 2 (always include DAPI and polyT) and can include 6 more. | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - -
diff --git a/docs/assays/metadata/MIBI.md b/docs/assays/metadata/MIBI.md index 6c06a9b1..6e500c70 100644 --- a/docs/assays/metadata/MIBI.md +++ b/docs/assays/metadata/MIBI.md @@ -1,104 +1,86 @@ ---- -layout: page ---- -# MIBI - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 2 (latest) - -## Version 2 (latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | The number of distinct color channels in the image. | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| roi_description | Textfield | A description of the anatomical structure being captured in the region of interest (ROI). | | True | -| roi_id | Numeric | Multiple images are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The ROI ID is a number from 1 to N representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the "Acquisition ID" and the "ROI ID" indicate the slide-ROI represented in the image. | | True | -| area_normalized_ion_dose_value | Numeric | Number of primary ions delivered to the sample per unit area. | | True | -| area_normalized_ion_dose_unit | Allowable Value | Area normalized ion dose unit. | ```nA*hr/mm2``` | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes. | | True | -| pixel_dwell_time_value | Numeric | Resident time of primary ion beam on each pixel to ionize it. | | True | -| pixel_dwell_time_unit | Allowable Value | Pixel dwell time unit. | ```ms``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - -
- - -
Version 1 - -## Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|--------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['mass_spectrometry_imaging'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['MIBI'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['protein'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| resolution_x_value | Numeric | The width of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_x_unit | Allowable Value | The unit of measurement of width of a pixel.(nm) | ['mm', 'um', 'nm'] | False | -| resolution_y_value | Numeric | The height of a pixel. (Akoya pixel is 377nm square) | | True | -| resolution_y_unit | Allowable Value | The unit of measurement of height of a pixel. (nm) | ['mm', 'um', 'nm'] | False | -| max_x_width_value | Numeric | Image width value of the ROI acquisition | | True | -| max_x_width_unit | Allowable Value | Units of image width of the ROI acquisition | ['um'] | False | -| max_y_height_value | Numeric | Image height value of the ROI acquisition | | True | -| max_y_height_unit | Allowable Value | Units of image height of the ROI acquisition | ['um'] | False | -| roi_description | Textfield | A description of the region of interest (ROI) captured in the image. | | True | -| roi_id | Numeric | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | True | -| acquisition_id | Textfield | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | True | -| area_normalized_ion_dose_unit | Allowable Value | Area normalized ion dose unit | ['nA*hr/mm2'] | False | -| area_normalized_ion_dose_value | Numeric | Number of primary ions delivered to the sample per unit area | | True | -| data_precision_bytes | Numeric | Numerical data precision in bytes | | True | -| dual_count_start | Numeric | Threshold for dual counting. | | True | -| end_datetime | Datetime | Time stamp indicating end of ablation for ROI | | True | -| pixel_dwell_time_value | Numeric | Resident time of primary ion beam on each pixel. | | True | -| pixel_dwell_time_unit | Allowable Value | Pixel dwell time unit. | ['ms'] | False | -| pixel_size_x_value | Numeric | Width value of the pixel or voxel measurement (distinct from the image resolution_x_value). | | True | -| pixel_size_x_unit | Allowable Value | Width unit of the pixel or voxel measurement. | ['nm'] | False | -| pixel_size_y_value | Numeric | Length value of the pixel or voxel measurement (distinct from the image resolution_y_value). | | True | -| pixel_size_y_unit | Allowable Value | Length unit of the pixel or voxel measurement. | ['nm'] | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare the sample for the assay. | ['Custom', 'Ionpath'] | True | -| preparation_instrument_model | Allowable Value | The model number/name of the instrument used to prepare the sample for the assay | ['Custom', 'MIBIscope 1', 'MIBIscope 2'] | True | -| primary_ion | Allowable Value | Primary ion. | ['Xe'] | True | -| primary_ion_current_value | Numeric | Primary ion current value. | | True | -| primary_ion_current_unit | Allowable Value | Primary ion current unit, typically nA or pA | ['nA', 'pA'] | False | -| reagent_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | True | -| section_prep_protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for preparing tissue sections for the assay. | | True | -| segment_data_format | Allowable Value | This refers to the data type, which is a "float" for the IMC counts. | ['float', 'integer', 'string'] | True | -| signal_type | Allowable Value | Type of signal measured per channel (usually dual counts) | ['dual count', 'pulse count', 'intensity value'] | True | -| start_datetime | Datetime | Time stamp indicating start of ablation for ROI | | True | -| antibodies_path | Textfield | Relative path to file with antibody information for this dataset. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
+--- +layout: page-triary +--- + +# MIBI Metadata Attributes + +Fields that are collected for MIBI data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path *| | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| number_of_antibodies *| | Number of antibodies | | +| number_of_channels *| | Number of fluorescent channels imaged during each cycle. | | +| slide_id *| | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| roi_description *| | A description of the region of interest (ROI) captured in the image. | | +| roi_id *| | Multiple images (1-n) are acquired from regions of interest (ROI1, ROI2, ROI3, etc) on a slide. The roi_id is a number from 1-n representing the ROI captured on a slide. | | +| acquisition_id *| | The acquisition_id refers to the directory containing the ROI images for a slide. Together, the acquisition_id and the roi_id indicate the slide-ROI represented in the image. | | +| area_normalized_ion_dose_value *| | Number of primary ions delivered to the sample per unit area | | +| area_normalized_ion_dose_unit *| | Area normalized ion dose unit | ```nA*hr/mm2``` | +| data_precision_bytes *| | Numerical data precision in bytes | | +| pixel_dwell_time_value *| | Resident time of primary ion beam on each pixel. | | +| pixel_dwell_time_unit *| | Pixel dwell time unit. | ```ms``` | +| antibodies_path *| | Relative path to file with antibody information for this dataset. | | +| primary_ion *| | Primary ion. | ```Xe``` | +| primary_ion_current_unit | | Primary ion current unit, typically nA or pA | ```nA``` ```pA``` | +| primary_ion_current_value *| | Primary ion current value. | | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ```sequence``` | +| description | | Free-text description of this assay. | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| protocol_io_doi | | | | +| reagent_prep_protocols_io_doi | | DOI for protocols.io referring to the protocol for preparing reagents for the assay. | | +| preparation_instrument_model | | The model number/name of the instrument used to prepare the sample for the assay | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare the sample for the assay. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| segment_data_format | | This refers to the data type, which is a "float" for the IMC counts. | ```float``` ```integer``` ```string``` | +| signal_type | | Type of signal measured per channel (usually dual counts) | ```dual count``` ```pulse count``` ```intensity value``` | +| dual_count_start | | Threshold for dual counting. | | +| start_datetime | | Time stamp indicating start of ablation for ROI | | +| end_datetime | | Time stamp indicating end of ablation for ROI | | +| resolution_x_unit | | The unit of measurement of width of a pixel.(nm) | ```mm``` ```um``` ```nm``` | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_unit | | The unit of measurement of height of a pixel. (nm) | ```mm``` ```um``` ```nm``` | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| max_x_width_unit | | Units of image width of the ROI acquisition | ```um``` | +| max_x_width_value | | Image width value of the ROI acquisition | | +| max_y_height_unit | | Units of image height of the ROI acquisition | ```um``` | +| max_y_height_value | | Image height value of the ROI acquisition | | +| pixel_size_x_unit | | Width unit of the pixel or voxel measurement. | ```nm``` | +| pixel_size_x_value | | Width value of the pixel or voxel measurement (distinct from the image resolution_x_value). | | +| pixel_size_y_unit | | Length unit of the pixel or voxel measurement. | ```nm``` | +| pixel_size_y_value | | Length value of the pixel or voxel measurement (distinct from the image resolution_y_value). | | +| resolution_x_value | | The width of a pixel. (Akoya pixel is 377nm square) | | +| resolution_y_value | | The height of a pixel. (Akoya pixel is 377nm square) | | +| version | | Version of the schema to use when validating this metadata. | ```1``` | diff --git a/docs/assays/metadata/MPLEx.md b/docs/assays/metadata/MPLEx.md deleted file mode 100644 index d7c3f0b7..00000000 --- a/docs/assays/metadata/MPLEx.md +++ /dev/null @@ -1,59 +0,0 @@ ---- -layout: page ---- -# MPLEx - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Assigned Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode```, ```Positive ion mode```, ```Negative ion mode``` | True | -| mass_to_charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass_to_charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | False | -| mass_to_charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | False | -| ion_mobility | Assigned Value | Specifies which technology was used for ion mobility spectrometry. Technologies for measuring ion mobility: Traveling Wave Ion Mobility Spectrometry (TWIMS), Trapped Ion Mobility Spectrometry (TIMS), High Field Asymmetric waveform ion Mobility Spectrometry (FAIMS), Drift Tube Ion Mobility Spectrometry (DTIMS), Structures for Lossless Ion Manipulations (SLIM), and cyclic Ion Mobility Spectrometry (cIMS). | ```TIMS```, ```SLIM```, ```FAIMS```, ```DTIMS```, ```cIMS```, ```TWIMS``` | False | -| ms_ionization_technique | Assigned Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```MALDI```, ```SIMS-C60```, ```LDI```, ```HESI```, ```nanoDESI```, ```MALDI-2```, ```DESI```, ```LA```, ```SIMS-H20```, ```ESI``` | True | -| ms_scan_mode | Assigned Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS2```, ```MS1```, ```MS3``` | True | -| label_name | Textfield | If the samples were labeled (e.g. TMT), provide the name/ID of the label on this sample. Leave blank if not applicable. | | False | -| lc_instrument_vendor | Assigned Value | The manufacturer of the instrument used for liquid chromatography. | ```Thermo Fisher Scientific```, ```Sciex```, ```In-House```, ```Agilent Technologies```, ```Waters```, ```Bruker```, ```Evosep``` | False | -| lc_instrument_model | Textfield | The model number/name of the instrument used for liquid chromatography. | | False | -| lc_column_model | Textfield | The model number/name of the liquid chromatography column. If it is a custom self-packed, pulled tip capillary is used enter “Pulled tip capilary”. | | False | -| lc_resin | Textfield | Details of the resin used for liquid chromatography, including vendor, particle size, pore size. | | False | -| lc_column_length_value | Numeric | Liquid chromatography column length. | | False | -| lc_column_length_unit | Assigned Value | Units for liquid chromatography column length (typically cm). | ```um```, ```mm```, ```cm``` | False | -| lc_temperature_value | Numeric | Liquid chromatography temperature. | | False | -| lc_inner_diameter_value | Numeric | Liquid chromatography column inner diameter. | | False | -| lc_flow_rate_value | Numeric | Value of flow rate. | | False | -| lc_gradient_value | Numeric | Liquid chromatography gradient. | | False | -| lc_gradient_unit | Assigned Value | Unit for liquid chromatography gradient | ```minute``` | False | -| lc_mobile_phase_a | Textfield | Composition of mobile phase A. | | False | -| lc_mobile_phase_b | Textfield | | | False | -| spatial_sampling_technique | Assigned Value | | ```nanoSPLITS```, ```nanoPOTS```, ```LESA```, ```microPOTS```, ```LCM```, ```microLESA``` | False | -| spatial_sampling_target | Textfield | Specifies the cell-type or functional tissue unit (FTU) that is targeted in the spatial profiling experiment. Leave blank if data are generated in imaging mode without a specific target structure. | | False | -| analysis_protocol_doi | Link | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| data_collection_mode | Assigned Value | Mode of data collection in tandem MS assays. Either DDA (Data-dependent acquisition), DIA (Data-independent acquisition), SRM (multiple reaction monitoring), or PRM (parallel reaction monitoring). | ```DDA```, ```PRM```, ```DIA```, ```SRM``` | False | -| lc_column_vendor | Assigned Value | The manufacturer of the liquid chromatography column unless self-packed, pulled tip capillary is used. | ```Thermo Fisher Scientific```, ```In-House```, ```Waters```, ```Bruker```, ```Evosep```, ```IonOpticks``` | False | -| lc_temperature_unit | Assigned Value | | ```celsius``` | False | -| lc_inner_diameter_unit | Assigned Value | | ```um```, ```mm```, ```cm``` | False | -| lc_flow_rate_unit | Assigned Value | Units of flow rate. | ```nL/min```, ```mL/min``` | False | -| spatial_sampling_type | Assigned Value | Specifies whether or not the analysis was performed in a spatially targeted manner. Spatial profiling experiments target specific tissue foci but do not necessarily generate images. Spatial imaging expriments collect data from a regular array (pixels) that can be visualized as heat maps of ion intensity at each location (molecular images). Leave blank if data are derived from bulk analysis. | ```Imaging```, ```Profiling``` | False | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/testing/MUSIC-(CEDAR).md b/docs/assays/metadata/MUSIC-(CEDAR).md similarity index 100% rename from docs/assays/metadata/testing/MUSIC-(CEDAR).md rename to docs/assays/metadata/MUSIC-(CEDAR).md diff --git a/docs/assays/metadata/MUSIC.md b/docs/assays/metadata/MUSIC.md index 58c71288..48a650ac 100644 --- a/docs/assays/metadata/MUSIC.md +++ b/docs/assays/metadata/MUSIC.md @@ -1,58 +1,70 @@ ---- -layout: page ---- -# MUSIC - -
Current Metadata Attributes - -## Current Metadata Attributes - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```14-17,14,14``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | False | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | False | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | False | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```Not applicable``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,78``` ```10,48,86``` ```0,20-23,41-44``` ```Not applicable``` | True | -| amount_of_input_analyte_unit | Allowable Value | Units of amount of entity input to assay value | ```ug``` ```ng``` | False | - -
\ No newline at end of file +--- +layout: page-triary +--- + +# MUSIC Metadata Attributes + +Fields that are collected for MUSIC data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id *| | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi *| | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | +| dataset_type *| | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium``` | +| analyte_class *| | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA``` | +| is_targeted *| | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | | +| acquisition_instrument_vendor *| | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | +| acquisition_instrument_model *| | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | +| source_storage_duration_value *| | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit *| | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | +| contributors_path *| | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | +| data_path *| | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | +| barcode_offset *| | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | +| barcode_read *| | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | +| barcode_size *| | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | +| umi_offset *| | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | +| umi_read *| | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | +| umi_size *| | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | +| assay_input_entity *| | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | +| number_of_input_cells_or_nuclei | | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | +| amount_of_input_analyte_value | | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | +| amount_of_input_analyte_unit | | Units of amount of entity input to assay value | ```ug``` ```ng``` | +| library_adapter_sequence *| | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | +| library_average_fragment_size *| | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | +| library_input_amount_value | | The amount of cDNA, after amplification, that was used for library construction. | | +| library_input_amount_unit | | unit of library input amount value | ```ng``` ```ul``` | +| library_output_amount_value | | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | +| library_output_amount_unit | | Units of library final yield. | ```ng``` ```ul``` | +| library_concentration_value *| | The concentration value of the pooled library samples submitted for sequencing. | | +| library_concentration_unit *| | Unit of library concentration value. | ```ng/ul``` ```nM``` | +| library_layout *| | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | +| library_preparation_kit *| | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```1 slides``` ```4 reactions; PN 1000338``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 reactions; PN 1000187``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | +| sample_indexing_kit *| | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001``` | +| sample_indexing_set | | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | +| is_technical_replicate *| | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | | +| expected_entity_capture_count | | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | +| sequencing_reagent_kit *| | Reagent kit used for sequencing | ```Custom``` ```Illumina``` ```HiSeq 3000/4000 PE Cluster Kit PE-410-1001``` ```PN 1000283, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles)``` ```PN 20046811, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles)``` ```PN 20046812, Illumina``` ```NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles)``` ```PN 20046813, Illumina``` ```NextSeq 2000 P3 Reagent Kit (300 Cycles)``` ```PN 20040561, Illumina``` ```NextSeq 2000 P3 Reagents Kit (100 Cycles)``` ```PN 20040559, Illumina``` ```NextSeq 500/550 Hi Output Kit 150 Cycles``` ```v2.5``` ```PN 20024907, Illumina``` ```NextSeq 500/550 Hi Output Kit 75 Cycles v2.5``` ```PN 20024906, Illumina``` ```NextSeq 500/550 Mid Output Kit 150 Cycles v2.5``` ```PN 20024904, Illumina``` ```NovaSeq 6000 S1 Reagent Kit (200 Cycles)``` ```PN 20012864, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028319, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028318, Illumina``` ```NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles)``` ```PN 20028317, Illumina``` ```NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles)``` ```PN 20028316, Illumina``` ```NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles)``` ```PN 20028312, Illumina``` ```NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles)``` ```PN 20028313, Illumina``` ```NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles)``` ```PN 20028401, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (100 Cycle)``` ```PN 20104703, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (200 Cycle)``` ```PN 20104704, Illumina``` ```NovaSeq X Series 1.5B Reagent Kit (300 Cycle)``` ```PN 20104705, Illumina``` ```NovaSeq X Series 10B Reagent Kit (100 Cycle)``` ```PN 20085596, Illumina``` ```NovaSeq X Series 10B Reagent Kit (200 Cycle)``` ```PN 20085595, Illumina``` ```NovaSeq X Series 10B Reagent Kit (300 Cycle)``` ```PN 20085594``` | +| sequencing_read_format *| | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | +| sequencing_batch_id | | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| capture_batch_id | | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist``` | +| preparation_instrument_kit | | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | +| metadata_schema_id *| | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes + + These attributes were supported by older metadata version specifications. They are no longer collected but there may be some older datasets that contain data for these attributes. + + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| diff --git a/docs/assays/metadata/Olink.md b/docs/assays/metadata/Olink.md deleted file mode 100644 index 2a36ed5b..00000000 --- a/docs/assays/metadata/Olink.md +++ /dev/null @@ -1,28 +0,0 @@ ---- -layout: page ---- -# Olink - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/PhenoCycler.md b/docs/assays/metadata/PhenoCycler.md deleted file mode 100644 index 7883ef67..00000000 --- a/docs/assays/metadata/PhenoCycler.md +++ /dev/null @@ -1,36 +0,0 @@ ---- -layout: page ---- -# PhenoCycler - -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| number_of_antibodies | Numeric | Number of antibodies | | True | -| number_of_channels | Numeric | Number of fluorescent channels imaged during each cycle. | | True | -| number_of_biomarker_imaging_rounds | Numeric | Number of imaging rounds to capture the tagged biomarkers. For CODEX a biomarker imaging round consists of 1. oligo application, 2. fluor application, 3. washes. For Cell DIVE a biomarker imaging round consists of 1. staining of a biomarker via secondary detection or direct conjugate and 2. dye inactivation. | | True | -| number_of_total_imaging_rounds | Numeric | The total number of acquisitions performed on microscope to collect autofluorescence/background or stained signal (e.g., histology). | | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| total_run_time_value | Numeric | How long the tissue was on the acquisition instrument. | | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| total_run_time_unit | Allowable Value | The units for the total run time unit field. | ```Hour``` ```Minute``` | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| nuclear_marker_or_stain | Allowable Value | For markers, an antibody-targetted molecule present in or around the cell nucleus, the protein or gene symbol that identifies the antibody target that is used as the nuclear marker. This symbol must match the antibody target that is either generated from the panel used or entered with custom panels. Preferably, if using a custom antibody marker, this symbol should be the HGNC symbol (https://www.genenames.org/). For non-protein targets this is the stain name (e.g., DAPI) and, when appropriate, associated staining kit and vendor. For the PhenoCycler, this symbol must match the value found in the XPD output file. | ```DAPI``` ```Not applicable``` | True | -| cell_boundary_marker_or_stain | Allowable Value | If a marker or stain was used to identify all cell boundaries in the tissue, then the name of the marker or stain should be included here. The name of the antibody-targeted molecule marker or non-antibody targeted molecule stain included here must be identical to what is found in the imaging data. For example, with the PhenoCycler, this name must match the value found in the XPD output file. If multiple marker or stains are used to identify all cell boundaries, then a comma separated list should be used here. | ```NAKATPASE``` ```CD298``` ```Not applicable``` | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | diff --git a/docs/assays/metadata/Pixel-seqV2.md b/docs/assays/metadata/Pixel-seqV2.md deleted file mode 100644 index 44e5a345..00000000 --- a/docs/assays/metadata/Pixel-seqV2.md +++ /dev/null @@ -1,37 +0,0 @@ ---- -layout: page ---- -# Pixel-seqV2 - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, this is the area of the FOV (aka ROI) region being captured. For Xenium this is the total area of the FOV regions (aka ROI) being captured. For Stereo-Seq this is the number of beads. | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2```, ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Assigned Value | The unit for spot size value. | ```um^2```, ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | True | -| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | True | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| number_of_additional_stains | Numeric | This would be minimally 2 (always include DAPI and polyT) and can include 6 more. | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/testing/RNAseq-(with-probes).md b/docs/assays/metadata/RNAseq-(with-probes).md similarity index 100% rename from docs/assays/metadata/testing/RNAseq-(with-probes).md rename to docs/assays/metadata/RNAseq-(with-probes).md diff --git a/docs/assays/metadata/RNAseq.md b/docs/assays/metadata/RNAseq.md index 8a07abc0..0592557c 100644 --- a/docs/assays/metadata/RNAseq.md +++ b/docs/assays/metadata/RNAseq.md @@ -1,425 +1,95 @@ ---- -layout: page ---- -# RNAseq - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
RNAseq Version 5 (current) - -## RNAseq Version 5 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | True | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | True | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | -| number_of_iterations_of_cdna_amplification | Numeric | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,78``` ```10,48,86``` ```Not applicable``` | True | -| amount_of_input_analyte_unit | Allowable Value | Units of amount of entity input to assay value | ```ug``` ```ng``` | False | - -
- -
RNAseq Version 2 - -## RNAseq Version 2 - -| Attribute | Type | Description | Allowable Values | required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | True | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | True | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | True | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| True | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, "Yes" or "No". If "Yes", FASTQ files in dataset need to be merged. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | -| amount_of_input_analyte_unit | Textfield | Units of amount of entity input to assay value | | False | -| number_of_iterations_of_cdna_amplification | Numeric | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,78``` ```10,48,86``` ```Not applicable``` | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```36``` ```Not applicable``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq (bulk)``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```PhenoCycler``` ```RNAseq (bulk)``` ```scATACseq``` ```scRNAseq``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```snATACseq``` ```snRNAseq``` ```Thick section Multiphoton MxIF``` ```Visium``` ```Xenium``` | True | - -
- -
bulk-RNA Version 1 - -## bulk-RNA Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['bulkATACseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| bulk_transposition_input_number_nuclei | Textfield | A number (no comma separators) | | True | -| bulk_atac_cell_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How was tissue stored and processed for cell/nuclei isolation | | True | -| is_technical_replicate | Allowable Value | Is this a sequencing replicate? | ['Yes','No']] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library_concentration_value | ['nM'] | False | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_creation_date | Datetime | date and time of library creation. YYYY-MM-DD, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s. | | False | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_pcr_cycles | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. Usually, this includes 5 pre-amplificationn cycles followed by 0-5 additional cycles determined by qPCR. | | True | -| library_preparation_kit | Textfield | Reagent kit used for library preparation | | True | -| sample_quality_metric | Textfield | This is a quality metric by visual inspection. This should answerthe question: Are the nuclei intact and are the nuclei free of significant amountsof debris? This can be captured at a high level, “OK” or “notOK”. | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| transposition_kit_number | Textfield | If Tn5 came from a kit, provide the catalog number. | | False | -| transposition_method | Textfield | Modality of capturing accessible chromatin molecules. The kit used, for example. | | True | -| transposition_transposase_source | Textfield | The source of the Tn5 transposase and transposon used for capturing accessible chromatin. | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
bulk-RNA Version 0 - -## bulk-RNA Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['bulk-RNA'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| bulk_rna_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How was tissue stored and processed for RNA isolation RNA_isolation_protocols_io_doi | | True | -| bulk_rna_yield_value | Numeric | RNA (ng) per Weight of Tissue (mg). Answer the question: How much RNA in ng was isolated? How much tissue in mg was initially used for isolating RNA? Calculate the yield by dividing total RNA isolated by amount of tissue used to isolate RNA from (ng/mg). | | True | -| bulk_rna_yield_units_per_tissue_unit | Allowable Value | RNA amount per Tissue input amount. Valid values should be weight/weight (ng/mg). | ['ng/mg'] | True | -| bulk_rna_isolation_quality_metric_value | Numeric | RIN value | | True | -| rnaseq_assay_input_value | Numeric | RNA input amount value to the assay | | True | -| rnaseq_assay_input_unit | Allowable Value | Units of RNA input amount to the assay | ['ug'] | False | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming. | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
scRNAseq Version 3 - -## scRNAseq Version 3 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['3'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The UMI sequence length in the 10xGenomics-v2 kit is 10 base pairs and the length in the 10xGenomics-v3 kit is 12 base pairs. | ['scRNAseq-10xGenomics-v2', 'scRNAseq-10xGenomics-v3', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, TRUE or FALSE | ['Yes','No']] | True | -| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). | | False | -| umi_read | Textfield | Which read file(s) contains the UMI (unique molecular identifier) barcode. | | True | -| umi_offset | Numeric | Position in the read at which the umi barcode starts. | | True | -| umi_size | Numeric | Length of the umi barcode in base pairs. | | True | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | False | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | False | -| expected_cell_count | Numeric | How many cells are expected? This may be used in downstream pipelines to guide selection of cell barcodes or segmentation parameters. | | False | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
scRNAseq Version 2 - -## scRNAseq Version 2 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics-v2', 'scRNAseq-10xGenomics-v3', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, TRUE or FALSE | ['Yes','No']] | True | -| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). | | False | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | False | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | False | -| expected_cell_count | Numeric | How many cells are expected? This may be used in downstream pipelines to guide selection of cell barcodes or segmentation parameters. | | False | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
scRNAseq Version 1 - -## scRNAseq Version 1 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in repliucate, TRUE or FALSE | ['Yes','No']] | True | -| cell_barcode_read | Textfield | Which read file contains the cell barcode | | True | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | True | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | True | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
scRNAseq Version 0 - -## scRNAseq Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['2'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['scRNAseq-10xGenomics-v2', 'scRNAseq-10xGenomics-v3', 'snRNAseq-10xGenomics-v2', 'snRNAseq-10xGenomics-v3', 'scRNAseq', 'sciRNAseq', 'snRNAseq', 'SNARE2-RNAseq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No']] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| sc_isolation_protocols_io_doi | Textfield | Textfield to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | True | -| sc_isolation_entity | Allowable Value | The type of single cell entity derived from isolation protocol | ['whole cell', 'nucleus', 'cell-cell multimer', 'spatially encoded cell barcoding'] | True | -| sc_isolation_tissue_dissociation | Textfield | The method by which tissues are dissociated into single cells in suspension. | | True | -| sc_isolation_enrichment | Allowable Value | The method by which specific cell populations are sorted or enriched. | ['none', 'FACS'] | True | -| sc_isolation_quality_metric | Textfield | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | True | -| sc_isolation_cell_number | Numeric | Total number of cell/nuclei yielded post dissociation and enrichment | | True | -| rnaseq_assay_input | Numeric | Number of cell/nuclei input to the assay | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| library_id | Textfield | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in replicate, TRUE or FALSE | ['Yes','No']] | True | -| cell_barcode_read | Textfield | Which read file(s) contains the cell barcode. Multiple cell_barcode_read files must be provided as a comma-delimited list (e.g. file1,file2,file3). | | False | -| cell_barcode_offset | Textfield | Position(s) in the read at which the cell barcode starts. | | False | -| cell_barcode_size | Textfield | Length of the cell barcode in base pairs | | False | -| expected_cell_count | Numeric | How many cells are expected? This may be used in downstream pipelines to guide selection of cell barcodes or segmentation parameters. | | False | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
\ No newline at end of file +--- +layout: page-triary +--- + +# RNAseq Metadata Attributes + +Fields that are collected for RNAseq data, available at ```dataset.metadata.``` +  + +* indicates a required field + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| parent_sample_id | | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | +| lab_id | | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | +| preparation_protocol_doi | | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | ```https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1``` | +| dataset_type | | The specific type of dataset being produced. | | +| analyte_class | | Analytes are the target molecules being measured with the assay. | | +| is_targeted | | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | | +| acquisition_instrument_vendor | | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | +| acquisition_instrument_model | | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | +| source_storage_duration_value | | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | +| source_storage_duration_unit | | The time duration unit of measurement | | +| time_since_acquisition_instrument_calibration_value | | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | +| time_since_acquisition_instrument_calibration_unit | | The time unit of measurement | | +| contributors_path | | Relative path to file with ORCID IDs for contributors for this dataset. | | +| data_path | | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | +| barcode_offset | | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | | +| barcode_read | | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | | +| barcode_size | | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | | +| umi_offset | | Position in the read at which the umi barcode starts. | | +| umi_read | | Which read file(s) contains the UMI (unique molecular identifier) barcode. | | +| umi_size | | Length of the umi barcode in base pairs. | | +| assay_input_entity | | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | | +| number_of_input_cells_or_nuclei | | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | +| amount_of_input_analyte_value | | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | +| amount_of_input_analyte_unit | | Units of amount of entity input to assay value | | +| library_adapter_sequence | | Adapter sequence to be used for adapter trimming | | +| library_average_fragment_size | | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | +| library_input_amount_value | | The amount of cDNA, after amplification, that was used for library construction. | | +| library_input_amount_unit | | unit of library input amount value | | +| library_output_amount_value | | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | +| library_output_amount_unit | | Units of library final yield. | | +| library_concentration_value | | The concentration value of the pooled library samples submitted for sequencing. | | +| library_concentration_unit | | Unit of library concentration value. | | +| library_layout | | State whether the library was generated for single-end or paired end sequencing. | | +| number_of_iterations_of_cdna_amplification | | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | +| number_of_pcr_cycles_for_indexing | | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | +| library_preparation_kit | | Reagent kit used for library preparation | | +| sample_indexing_kit | | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | | +| sample_indexing_set | | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | +| is_technical_replicate | | Is the sequencing reaction run in replicate, TRUE or FALSE | | +| expected_entity_capture_count | | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | +| sequencing_reagent_kit | | Reagent kit used for sequencing | | +| sequencing_read_format | | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | +| sequencing_batch_id | | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| capture_batch_id | | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | +| preparation_instrument_vendor | | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | | +| preparation_instrument_model | | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | +| preparation_instrument_kit | | The reagent kit used with the preparation instrument. | | +| metadata_schema_id | | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | + + +  + +## Deprecated Attributes +  + + indicates a field that was previously required + +| Attribute | Type | Description | Allowable Values | +|------|------|-------------|-------------------| +| assay_category | | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | | +| bulk_rna_isolation_protocols_io_doi | | Link to a protocols document answering the question: How was tissue stored and processed for RNA isolation RNA_isolation_protocols_io_doi | | +| bulk_rna_isolation_quality_metric_value | | RIN value | | +| bulk_rna_yield_units_per_tissue_unit | | RNA amount per Tissue input amount. Valid values should be weight/weight (ng/mg). | | +| bulk_rna_yield_value | | RNA (ng) per Weight of Tissue (mg). Answer the question: How much RNA in ng was isolated? How much tissue in mg was initially used for isolating RNA? Calculate the yield by dividing total RNA isolated by amount of tissue used to isolate RNA from (ng/mg). | | +| donor_id | | HuBMAP Display ID of the donor of the assayed tissue. | | +| execution_datetime | | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | +| library_construction_protocols_io_doi | | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | +| library_id | | A library ID, unique within a TMC, which allows corresponding RNA and chromatin accessibility datasets to be linked. | | +| operator | | Name of the person responsible for executing the assay. | | +| operator_email | | Email address for the operator. | | +| pi | | Name of the principal investigator responsible for the data. | | +| pi_email | | Email address for the principal investigator. | | +| rnaseq_assay_method | | The kit used for the RNA sequencing assay | | +| sc_isolation_enrichment | | The method by which specific cell populations are sorted or enriched. | | +| sc_isolation_protocols_io_doi | | Link to a protocols document answering the question: How were single cells separated into a single-cell suspension? | | +| sc_isolation_quality_metric | | A quality metric by visual inspection prior to cell lysis or defined by known parameters such as wells with several cells or no cells. This can be captured at a high level. | | +| sc_isolation_tissue_dissociation | | The method by which tissues are dissociated into single cells in suspension. | | +| sc_isolation_cell_number | | Total number of cell/nuclei yielded post dissociation and enrichment | | +| sequencing_phix_percent | | Percent PhiX loaded to the run | | +| sequencing_read_percent_q30 | | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | +| version | | Version of the schema to use when validating this metadata. | | +| description | | Free-text description of this assay. | | diff --git a/docs/assays/metadata/RNAseqWithProbes.md b/docs/assays/metadata/RNAseqWithProbes.md deleted file mode 100644 index fb8e9742..00000000 --- a/docs/assays/metadata/RNAseqWithProbes.md +++ /dev/null @@ -1,63 +0,0 @@ ---- -layout: page ---- -# RNAseq-(with-probes) -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| barcode_read | Allowable Value | Which read file contains the cell or capture spot barcode. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| barcode_size | Allowable Value | Length of the cell or capture spot barcode in base pairs. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences, the offsets. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if the source material is barcoded. This field is used to determine which analysis pipeline to run. | ```14``` ```16``` ```40``` ```8,8,8``` ```8,6``` ```8,8``` ```Not applicable``` | True | -| umi_read | Allowable Value | Which read file contains the UMI barcode. This should be included when constructing sequencing libraries with a non-commercial kit. | ```Read 2 (R2)``` ```Read 1 (R1)``` ```Not applicable``` | True | -| umi_size | Allowable Value | Length of the umi barcode in base pairs. This should be included when constructing sequencing libraries with a non-commercial kit. This field is required if UMI are present. This field is used to determine which analysis pipeline to run. | ```8``` ```9``` ```10``` ```12``` ```14``` ```Not applicable``` | True | -| assay_input_entity | Allowable Value | This is the entity from which the analyte is being captured. For example, for bulk sequencing this would be "tissue", while it would be "single cell" for single cell sequencing. This field is used to determine which analysis pipeline to run. | ```area of interest``` ```single cell``` ```single nucleus``` ```spot``` ```tissue (bulk)``` | True | -| number_of_input_cells_or_nuclei | Numeric | How many cells or nuclei were input to the assay? This is typically not available for preparations working with bulk tissue. | | False | -| library_adapter_sequence | Textfield | 5’ and/or 3’ read adapter sequences used as part of the library preparation protocol to render the library compatible with the sequencing protocol and instrumentation. This should be provided as comma-separated list of key:value pairs (adapter name:sequence). | | True | -| library_average_fragment_size | Numeric | Average size of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. Numeric value in base pairs (bp). | | True | -| library_input_amount_value | Numeric | The amount of cDNA, after amplification, that was used for library construction. | | False | -| library_input_amount_unit | Allowable Value | unit of library input amount value | ```ng``` ```ul``` | False | -| library_output_amount_value | Numeric | Total amount (eg. nanograms) of library after the clean-up step of final pcr amplification step. Answer the question: What is the Qubit measured concentration (ng/ul) times the elution volume (ul) after the final clean-up step? | | False | -| library_output_amount_unit | Allowable Value | Units of library final yield. | ```ng``` ```ul``` | False | -| library_concentration_value | Numeric | The concentration value of the pooled library samples submitted for sequencing. | | True | -| library_concentration_unit | Allowable Value | Unit of library concentration value. | ```ng/ul``` ```nM``` | True | -| library_layout | Allowable Value | Whether the library was generated for single-end or paired end sequencing | ```paired-end``` ```single-end``` | True | -| number_of_pcr_cycles_for_indexing | Numeric | Number of PCR cycles performed in order to add adapters and amplify the library. This does not include the cDNA amplification which is captured in the "number of iterations of cDNA amplification" field. | | True | -| library_preparation_kit | Allowable Value | Reagent kit used for library preparation | ```10X Genomics; Automated Library Construction Kit``` ```24 rxns; PN 1000428``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```24 rxns; PN 1000290``` ```10X Genomics; Chromium Next GEM Automated Single Cell 5' Kit v2``` ```4 rxns; PN 1000298``` ```10X Genomics; Chromium Next GEM Single Cell 3' GEM``` ```Library & Gel Bead Kit v3.1``` ```16 rxns; PN 1000121``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```48 rxns; PN 1000348``` ```10X Genomics; Chromium Next GEM Single Cell 3' HT Kit v3.1``` ```8 rxns; PN 1000370``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```16 rxns; PN 1000268``` ```10X Genomics; Chromium Next GEM Single Cell 3' Kit v3.1``` ```4 rxns; PN 1000269``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```16 rxns; PN 1000263``` ```10X Genomics; Chromium Next GEM Single Cell 5' Kit v2``` ```4 rxns; PN 1000265``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Hybridization & Library Kit``` ```4 rxns; PN 1000415``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Chromium Single Cell 3' GEM``` ```Library & Gel Bead Kit v3``` ```4 rxns PN 1000092``` ```10X Genomics; Chromium Single Cell 3' Library & Gel Bead Kit``` ```4 rxns; PN 120267``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```11 mm``` ```2 reactions; PN 1000522``` ```10X Genomics; Visium CytAssist Spatial Gene Expression for FFPE``` ```Human Transcriptome``` ```6.5mm``` ```4 reactions; PN 1000520``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Human Transcriptome``` ```1 slides``` ```4 reactions; PN 1000338``` ```10X Genomics; Visium Spatial for FFPE Gene Expression Kit``` ```Mouse Transcriptome``` ```4 rxns; PN 1000339``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```1 slides``` ```4 reactions; PN 1000187``` ```10X Genomics; Visium Spatial Gene Expression Slide and Reagent Kit``` ```4 slides``` ```16 reactions; PN 1000184``` ```Custom``` ```Illumina; TruSeq Stranded mRNA Library Prep (48 samples); PN 20020594``` ```Illumina; TruSeq Stranded mRNA Library Prep (96 samples); PN 20020595``` ```New England BioLabs; NEBNext Ultra II RNA Library Prep Kit for Illumina; PN E7770``` ```Parse Biosciences; Evercode WT Mini v2 Kit``` ```12 rxns; PN ECW02010``` ```Parse Biosciences; Evercode WT v2 Kit``` ```48 rxns; PN ECW02030)``` | False | -| sample_indexing_kit | Allowable Value | Indexes are needed for multiplexing sequencing libraries for simultaneous sequencing (pooling) and proper attachment to the Illumina flowcell. Each indexing kit would have a number of compatible sequences ("sample indexing sets") that are used to label some number of samples (the number of sets depend on the kit). | ```10X Genomics; Chromium i7 Sample Index Plate (96 rxn); PN 220103``` ```10X Genomics; Dual Index Kit TS``` ```Set A; PN 1000251``` ```10X Genomics; Dual Index Kit TT``` ```Set A (96 rxn); PN 1000215``` ```10X Genomics; Single Index Kit N``` ```Set A (96 rxn); PN 1000212``` ```Custom``` ```Illumina; IDT for Illumina - TruSeq RNA UD Indexes v2 (96 Indexes``` ```96 Samples); PN 20040871``` ```Illumina; TruSeq RNA CD Index Plate (96 Indexes``` ```96 Samples); PN 20019792``` ```Illumina; TruSeq RNA Single Indexes Set A (12 Indexes``` ```48 Samples); PN 20020492``` ```Illumina; TruSeq RNA Single Indexes Set B (12 Indexes``` ```48 Samples); PN 20020493``` ```Integrated DNA Technologies: Custom DNA Oligos``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-AB``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-CD``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-EF``` ```NanoString Technologies; GeoMx Seq Code Pack; PN GMX-NGS-SEQ-GH``` ```Not applicable``` ```Parse Biosciences; Fragmentation Reagents; PN WX100``` ```Parse Biosciences; UDI Plate - WT; PN UDI1001 ```| False | -| sample_indexing_set | Textfield | The specific sequencing barcode index set used, selected from the sample indexing kit. Example: For 10X this might be "SI-GA-A1", for Nextera "N505 - CTCCTTAC" | | False | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| expected_entity_capture_count | Numeric | Number of cells, nuclei or capture spots expected to be captured by the assay. For Visium this is the total number of spots covered by tissue, within the capture area. | | False | -| sequencing_reagent_kit | Allowable Value | Reagent kit used for sequencing |```Custom``` ```Illumina; HiSeq 3000/4000 PE Cluster Kit PE-410-1001; PN 1000283``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (100 Cycles); PN 20046811``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (200 Cycles); PN 20046812``` ```Illumina; NextSeq 1000/2000 P2 Reagent v3 Kit (300 Cycles); PN 20046813``` ```Illumina; NextSeq 2000 P3 Reagent Kit (300 Cycles); PN 20040561``` ```Illumina; NextSeq 2000 P3 Reagents Kit (100 Cycles); PN 20040559``` ```Illumina; NextSeq 500/550 Hi Output Kit 150 Cycles; v2.5; PN 20024907``` ```Illumina; NextSeq 500/550 Hi Output Kit 75 Cycles v2.5; PN 20024906``` ```Illumina; NextSeq 500/550 Mid Output Kit 150 Cycles v2.5; PN 20024904``` ```Illumina; NovaSeq 6000 S1 Reagent Kit (200 Cycles); PN 20012864``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (100 Cycles); PN 20028319``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (200 Cycles); PN 20028318``` ```Illumina; NovaSeq 6000 S1 Reagent v1.5 Kit (300 Cycles); PN 20028317``` ```Illumina; NovaSeq 6000 S2 Reagent v1.5 Kit (100 Cycles); PN 20028316``` ```Illumina; NovaSeq 6000 S4 Reagent Kit v1.5 (300 cycles); PN 20028312``` ```Illumina; NovaSeq 6000 S4 Reagent v1.5 Kit (200 Cycles); PN 20028313``` ```Illumina; NovaSeq 6000 SP Reagent v1.5 Kit (100 Cycles); PN 20028401``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (100 Cycle); PN 20104703``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (200 Cycle); PN 20104704``` ```Illumina; NovaSeq X Series 1.5B Reagent Kit (300 Cycle); PN 20104705``` ```Illumina; NovaSeq X Series 10B Reagent Kit (100 Cycle); PN 20085596``` ```Illumina; NovaSeq X Series 10B Reagent Kit (200 Cycle); PN 20085595``` ```Illumina; NovaSeq X Series 10B Reagent Kit (300 Cycle); PN 20085594``` | True | -| sequencing_read_format | Textfield | Number of sequencing cycles in each round of sequencing (i.e., Read1, i7 index, i5 index, and Read2). This is reported as a comma-delimited list. Example: For 10X snATAC-seq (R1,Index,R2,R3) this might be: 50,8,16,50. For SNARE-seq2 this might be: 75,94,8,75 | | True | -| sequencing_batch_id | Textfield | The ID for the sequencing run. This could, for example, be the chip ID and should allow users the ability to determine which samples were processed together in a sequencing run. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| capture_batch_id | Textfield | A lab-generated ID to identify which cells were captured at the same time. This would, for example, be an ID to denote which datasets were derived from a single 10X Genomics Chromium Controller run. In the case of the 10X Controller this could be the chip ID and would allow users the ability to determine which samples were processed together in a Chromium controller. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| amount_of_input_analyte_value | Numeric | The amount of RNA or DNA input to the assay, typically measured by a Qubit, BioAnalyzer, or TapeStation. In most single cell/nuclei assays, this value isn't available. | | False | -| number_of_iterations_of_cdna_amplification | Numeric | This is the amplification of the cDNA prior to library construction. This is typically a PCR amplification, while for linear amplification methods like aRNA this would be the number of rounds of aRNA. | | True | -| preparation_instrument_kit | Allowable Value | The reagent kit used with the preparation instrument. | ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```16 rxns; PN 1000127``` ```10X Genomics; Chromium Next GEM Chip G Single Cell Kit``` ```48 rxns; PN 1000120``` ```10X Genomics; Chromium Next GEM Chip K Automated Single Cell Kit``` ```48 rxns; PN 1000289``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```16 rxns; PN 1000287``` ```10X Genomics; Chromium Next GEM Chip K Single Cell Kit``` ```48 rxns; PN 1000286``` ```10X Genomics; Chromium Next GEM Chip Q Single Cell Kit``` ```16 rxns; PN 1000422``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```16 rxn; PN 1000283``` ```10X Genomics; Chromium NextGem Single Cell Multiome ATAC + Gene Expression Reagent Bundle``` ```4 rxn; PN 1000285``` ```10X Genomics; Visium FFPE Reagent Kit v2-Small``` ```PN 1000436``` ```Custom``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| umi_offset | Allowable Value | Position in the read at which the UMI barcode starts. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```16``` ```34``` ```36``` ```Not applicable``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| barcode_offset | Allowable Value | Positions in the read at which the cell or capture spot barcodes start. Cell and capture spot barcodes are, for example, 3 x 8 bp sequences that are spaced by constant sequences (the offsets). First barcode at position 0, then 38, then 76. This should be included when constructing sequencing libraries with a non-commercial kit. | ```0``` ```8``` ```20``` ```1,27``` ```0,38,76``` ```10,48,86``` ```Not applicable``` | True | -| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ```Yes``` ```No``` | True | -| probe_hybridization_time_value | Numeric | How long was the oligo-conjugated RNA or oligo-conjugated antibody probes hybridized with the sample? | | True | -| probe_hybridization_time_unit | Allowable Value | The units for probe hybridization time value. | ```Hour``` ```Minute``` | True | -| oligo_probe_panel | Allowable Value | This is the probe panel used to target genes and/or proteins. In cases where there is a core panel and add-on modules, the core panel should be selected here. If additional panels are used, then they must be included in the "additional_panels_used.csv" file that's uploaded with the dataset. | ```10x Genomics; Chromium Fixed RNA Kit``` ```Human Transcriptome``` ```4 rxns x 1 BC; PN 1000474``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```16 rxns; PN 1000420``` ```10X Genomics; Chromium Next GEM Single Cell Fixed RNA Human Transcriptome Probe Kit``` ```64 rxns; PN 1000456``` ```10x Genomics; Visium Human Transcriptome Probe Kit v2 - Small; PN 1000466``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Large; PN 1000364``` ```10x Genomics; Visium Human Transcriptome Probe Kit-Small; PN 1000363``` ```10x Genomics; Visium Mouse Transcriptome Probe Kit - Small; PN 1000365``` ```Custom``` ```NanoString Technologies; GeoMx Human Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-HuWTA-4``` ```NanoString Technologies; GeoMx Mouse Whole Transcriptome Atlas``` ```4 slides; PN GMX-RNA-NGS-MsWTA-4``` | True | -| is_custom_probes_used | Allowable Value | State ("Yes" or "No") whether custom RNA or antibody probes were used. If custom probes were used, they must be listed in the "custom_probe_set.csv" file. | ```Yes``` ```No``` | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| amount_of_input_analyte_unit | Allowable Value | Units of amount of entity input to assay value | ```ug``` ```ng``` | False | - -
diff --git a/docs/assays/metadata/Raman-Imaging.md b/docs/assays/metadata/Raman-Imaging.md deleted file mode 100644 index 95665aef..00000000 --- a/docs/assays/metadata/Raman-Imaging.md +++ /dev/null @@ -1,45 +0,0 @@ ---- -layout: page ---- -# Raman-Imaging - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| is_image_preprocessing_required | Radio | Indicates whether image preprocessing is necessary based on the type of acquisition instrument used, such as a microscope or slide scanner. This may involve steps like fusing image tiles to assemble the complete image. Example: Yes | ```Yes```, ```No``` | False | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | False | -| tiled_image_columns | Numeric | The number of columns used in the stitching process of a tiled image, often referred to as the grid size in the x-dimension. Example: 5 | | False | -| tiled_image_count | Numeric | The total number of raw tiled images captured, which are intended to be stitched together. Example: 75 | | False | -| intended_tile_overlap_percentage | Numeric | The intended percentage of overlap between tiled images. This value serves as the set point, although slight variations may occur during image acquisition due to stage registration. Example: 5 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq```, ```PhenoCycler``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| tile_configuration | Assigned Value | The configuration of tiles used for stitching in the assay process. If no tile configuration is applicable, enter "Not applicable". Example: Row-by-row | ```Column-by-column```, ```Not applicable```, ```Snake-by-columns```, ```Row-by-row```, ```Snake-by-rows``` | False | -| scan_direction | Assigned Value | The direction of imaging, which is necessary for the stitching process. Example: Left-and-down | ```Left-and-down```, ```Right-and-down```, ```Not applicable```, ```Right-and-up```, ```Left-and-up``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| number_of_pixels | Numeric | The total number of spatial sampling points in an image; for example, in a Raman image, each pixel corresponds to one recorded Raman spectrum. Example: 40000 | | True | -| pixel_physical_size_height_value | Numeric | The physical height of a single pixel in the image. Example: 1000 | | True | -| pixel_physical_size_height_unit | Assigned Value | The unit of measurement for the pixel physical size height value. If the pixel height is not specified, this field may be left blank. Example: um | ```um```, ```mm```, ```nm``` | True | -| pixel_physical_size_width_value | Numeric | The physical width of a single pixel in the image. Example: 1000 | | True | -| pixel_physical_size_width_unit | Assigned Value | The unit of measurement for the pixel physical size width value. If the pixel width value is not specified, this field may be left blank. Example: um | ```um```, ```mm```, ```nm``` | True | -| pixel_physical_size_depth_value | Numeric | The physical depth of a single pixel in the image. Example: 10 | | True | -| pixel_physical_size_depth_unit | Assigned Value | The unit of measurement for the pixel physical size depth value. If the pixel depth value is not specified, this field may be left blank. Example: um | ```um```, ```mm```, ```nm``` | True | -| objective_numerical_aperture | Numeric | Numerical aperture of the microscope objective used to focus the excitation laser on the sample and collect the resulting scattered signal, such as Raman-scattered light. Example: 0.5 | | True | -| laser_power | Numeric | Power of the excitation laser at the sample’s focal plane, measured after the objective and reported in milliwatts (mW). Example: 10 | | True | -| raman_shift_range | Textfield | Range of Raman shifts acquired in the measurement, expressed in wavenumbers (cm⁻¹). Example: 400-3200 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/SIMS.md b/docs/assays/metadata/SIMS.md deleted file mode 100644 index 3880a74d..00000000 --- a/docs/assays/metadata/SIMS.md +++ /dev/null @@ -1,39 +0,0 @@ ---- -layout: page ---- -# SIMS - -
Version 2 (latest) - -## Version 2 (latest) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mass_analysis_polarity | Allowable Value | The polarity of the mass analysis (positive or negative ion modes). | ```Negative and positive ion mode``` ```Negative ion mode``` ```Positive ion mode``` | True | -| mass_resolving_power | Numeric | The mass resolving power m/∆m, where ∆m is defined as the full width at half-maximum (FWHM) for a given peak with a specified mass-to-charge (m/z). (unitless) | | True | -| mass-to-charge_resolving_power | Numeric | The peak (m/z) used to calculate the resolving power. | | True | -| matrix_deposition_method | Allowable Value | Common methods of depositing matrix for assisting in desorption and ionization in imaging mass spectrometry include robotic spotting, electrospray deposition, and sublimation. | ```Electrospray deposition``` ```Not applicable``` ```Robotic spotting``` ```Robotic spraying``` ```Sublimation``` | False | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | -| preparation_matrix | Allowable Value | The matrix is a compound of crystallized molecules that acts like a buffer between the sample and the ionizing probe. It also helps ionize the sample, carrying it along the flight tube so it can be detected. | ```2,5-DHA (2,5-dihydroxyacetophenone)``` ```2,5-DHB (2,5-Dihydroxybenzoic acid)``` ```9-AA (9-aminoacridine)``` ```CHCA (alpha-cyano-4-hydroxy-cinnamic acid)``` ```DAN (1,5-diaminonapthalene)``` ```DMACA (4-(dimethylamino)cinnamic acid)``` ```NEDC (N-(1-naphthyl) ethylenediamine dihydrochloride)``` ```SA (sinapic acid)``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| mass-to-charge_range_low_value | Numeric | The low value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| mass-to-charge_range_high_value | Numeric | The high value of the scanned mass-to-charge range, for MS1. (unitless) | | False | -| analysis_protocol_doi | Textfield | A DOI to a protocols.io protocol describing the software and database(s) used to process the raw data. Example: https://dx.doi.org/10.17504/protocols.io.bsu5ney6 | | True | -| ms_ionization_technique | Allowable Value | The ionization approach (i.e., sample probing method) for performing imaging mass spectrometry. | ```DESI``` ```ESI``` ```HESI``` ```LA``` ```LDI``` ```MALDI``` ```MALDI-2``` ```nanoDESI``` ```SIMS-C60``` ```SIMS-H20 ```| True | -| ms_scan_mode | Allowable Value | MS (mass spectrometry) scan mode refers to the number of steps in the separation of fragments. | ```MS1``` ```MS2``` ```MS3``` | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
diff --git a/docs/assays/metadata/STARmap.md b/docs/assays/metadata/STARmap.md deleted file mode 100644 index 030cdade..00000000 --- a/docs/assays/metadata/STARmap.md +++ /dev/null @@ -1,43 +0,0 @@ ---- -layout: page ---- -# STARmap - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq```, ```PhenoCycler``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| mapped_area_value | Numeric | The mapped area value, which refers to the specific area covered or captured in various assays. For Visium, it is the area of spots covered by tissue within the captured area, excluding the total possible captured area. For GeoMx, it refers to the area of the AOI being captured. In HiFi, it is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, it indicates the area of the FOV (also known as ROI) region being captured. For Xenium, it is the total area of the FOV regions (also known as ROI) being captured. For Stereo-Seq, this value represents the number of beads. Example: 42.25 | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapped area value. If mapping area is not specified, this field may be left blank. Example: um^2 | ```um^2```, ```mm^2``` | True | -| target_retrieval_incubation_temperature | Numeric | The incubation temperature required for target retrieval, which is typically 100 degrees Celsius for RNA assays and 80 degrees Celsius for protein assays. Example: 100 | | False | -| target_retrieval_incubation_time_value | Numeric | The duration for which a sample is exposed to a target retrieval solution. Example: 15 | | False | -| target_retrieval_incubation_time_unit | Assigned Value | The unit of measurement for the target retrieval incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| proteinasek_concentration | Numeric | The concentration of the enzyme Proteinase K within a sample, measured in micrograms per milliliter (ug/ml). Example: 10 | | False | -| proteinasek_incubation_time_value | Numeric | The duration for which a sample is incubated with Proteinase K. Example: 15 | | False | -| proteinasek_incubation_time_unit | Assigned Value | The unit of measurement for the proteinaseK incubation time value. If no incubation time is specified, this field may be left blank. Example: minute | ```minute``` | False | -| probe_hybridization_time_value | Numeric | The duration for which the oligo-conjugated RNA or oligo-conjugated antibody probes were hybridized with the sample. Example: 30 | | False | -| probe_hybridization_time_unit | Assigned Value | The unit of measurement for the probe hybridization time value. If the hybridization time is not specified, this field may be left blank. Example: minute | ```hour```, ```minute``` | False | -| is_custom_probes_used | Radio | Indicates whether custom RNA or antibody probes were utilized in the assay. If custom probes were employed, they should be documented in the "custom_probe_set.csv" file. Example: No | ```Yes```, ```No``` | True | -| number_of_panel_targets | Numeric | The number of panel targets, which refers to the total count of genes, RNA isoforms, or RNA regions that are targeted by probes. Example: 1000 | | True | -| anatomical_structure_label | Textfield | The label for the overarching anatomical structure. If the anatomical structure is not applicable or not specified, this field may be left blank. Example: Kidney | | False | -| anatomical_structure_id | Textfield | The ontology ID associated with the anatomical structure, typically represented by an UBERON ID. Example: UBERON:0002113 | | False | -| non_global_files | Textfield | Specifies a semicolon-separated list of non-global files that are to be included in the dataset. The file paths assume that the files are located in the "TOP/non-global/" directory. For instance, if the file is located at TOP/non-global/lab_processed/images/1-tissue-boundary.geojson, the value for this field would be "./lab_processed/images/1-tissue-boundary.geojson". Once ingested, these files will be copied to their appropriate locations within the respective dataset directory tree. This field is intended for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee Example: ./lab_processed/images/1-tissue-boundary.geojson | | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/SecondHarmonicGeneration.md b/docs/assays/metadata/SecondHarmonicGeneration.md deleted file mode 100644 index 22b4169b..00000000 --- a/docs/assays/metadata/SecondHarmonicGeneration.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# SGH -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
diff --git a/docs/assays/metadata/Seq-Scope.md b/docs/assays/metadata/Seq-Scope.md deleted file mode 100644 index e8e33049..00000000 --- a/docs/assays/metadata/Seq-Scope.md +++ /dev/null @@ -1,39 +0,0 @@ ---- -layout: page ---- -# Seq-Scope - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Pixel-seq```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx```, ```MERFISH```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| analyte_class | Assigned Value | Analytes are the target molecules being measured with the assay. | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Q Exactive HF```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable```, ```Orbitrap Eclipse Tribrid```, ```MIBIscope```, ```IN Cell Analyzer 2200```, ```timsTOF FleX MALDI-2``` | True | -| source_storage_duration_value | Numeric | How long was the source material stored, prior to this sample being processed? For assays applied to tissue sections, this would be how long the tissue section (e.g., slide) was stored, prior to the assay beginning (e.g., imaging). For assays applied to suspensions such as sequencing, this would be how long the suspension was stored before library construction began. | | True | -| source_storage_duration_unit | Assigned Value | The time duration unit of measurement | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. | | False | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The time unit of measurement | ```month```, ```day```, ```year``` | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| is_targeted | Radio | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes,No``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, this is the area of the FOV (aka ROI) region being captured. For Xenium this is the total area of the FOV regions (aka ROI) being captured. For Stereo-Seq this is the number of beads. | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2```, ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Assigned Value | The unit for spot size value. | ```um^2```, ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Assigned Value | Units corresponding to inter-spot distance | ```um``` | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | True | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| number_of_additional_stains | Numeric | This would be minimally 2 (always include DAPI and polyT) and can include 6 more. | | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/Slide-seq.md b/docs/assays/metadata/Slide-seq.md deleted file mode 100644 index 3f7fad31..00000000 --- a/docs/assays/metadata/Slide-seq.md +++ /dev/null @@ -1,94 +0,0 @@ ---- -layout: page ---- -# SnareSeq2 - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 1 (no longer accepting data) - -## Version 1 (no longer accepting data) - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Slide-seq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes', 'No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| puck_id | Textfield | Slide-seq captures RNA sequence data on spatially barcoded arrays of beads. Beads are fixed to a slide in a region shaped like a round puck. Each puck has a unique puck_id. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in repliucate, TRUE or FALSE | ['Yes', 'No'] | True | -| bead_barcode_read | Textfield | Which read file contains the bead barcode | | True | -| bead_barcode_offset | Textfield | Position(s) in the read at which the bead barcode starts | | True | -| bead_barcode_size | Textfield | Length of the bead barcode in base pairs | | True | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['Slide-seq'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['RNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes', 'No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| rnaseq_assay_method | Textfield | The kit used for the RNA sequencing assay | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used, e.g. "Smart-Seq2", "Drop-Seq", "10X v3". | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | Adapter sequence to be used for adapter trimming | | True | -| puck_id | Textfield | Slide-seq captures RNA sequence data on spatially barcoded arrays of beads. Beads are fixed to a slide in a region shaped like a round puck. Each puck has a unique puck_id. | | True | -| is_technical_replicate | Allowable Value | Is the sequencing reaction run in repliucate, TRUE or FALSE | ['Yes', 'No'] | True | -| bead_barcode_read | Textfield | Which read file contains the bead barcode | | True | -| bead_barcode_offset | Textfield | Position(s) in the read at which the bead barcode starts | | True | -| bead_barcode_size | Textfield | Length of the bead barcode in base pairs | | True | -| library_pcr_cycles | Numeric | Number of PCR cycles to amplify cDNA | | True | -| library_pcr_cycles_for_sample_index | Numeric | Number of PCR cycles performed for library indexing | | True | -| library_final_yield_value | Numeric | Total number of ng of library after final pcr amplification step. This is the concentration (ng/ul) * volume (ul) | | True | -| library_final_yield_unit | Allowable Value | Units of final library yield | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/SnareSeq2.md b/docs/assays/metadata/SnareSeq2.md deleted file mode 100644 index 2ce1a047..00000000 --- a/docs/assays/metadata/SnareSeq2.md +++ /dev/null @@ -1,19 +0,0 @@ ---- -layout: page ---- -# SnareSeq2 -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|----------------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| number_of_pre-amplification_pcr_cycles | Numeric | The number of PCR cycles run after the Chromium Controller step and prior to separating the suspension and initiating library construction | | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/ThickSectionMultiphotonMxIF.md b/docs/assays/metadata/ThickSectionMultiphotonMxIF.md deleted file mode 100644 index a6238240..00000000 --- a/docs/assays/metadata/ThickSectionMultiphotonMxIF.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# MxIF -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-----------------------------------------------------|-----------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| source_storage_duration_value | Numeric | How long was the source material (parent) stored, prior to this sample being processed. | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The amount of time since the acqusition instrument was last serviced by the vendor. This provides a metric for assessing drift in data capture. | | False | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "/TEST001-RK/" for this field. If there are multiple directory levels, use the format "/TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| is_image_preprocessing_required | Allowable Value | Depending on if the acquisition instrument was a microscope, slide scanner, etc. will indicate whether or not any level of preprocessing was required to assemble the image (e.g., fusing image tiles) . | ```Yes``` ```No``` | False | -| slide_id | Textfield | A unique ID denoting the slide used. This allows users the ability to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name, to prevent values overlapping across centers. | | False | -| tiled_image_columns | Numeric | This is how many columns used in stitching. This is sometimes referred to as the grid size x. | | False | -| tiled_image_count | Numeric | This is the total number of raw (tiled) images captured, that are to be stitched together. | | False | -| intended_tile_overlap_percentage | Numeric | The amount of overlap between tiled images. This is the set point, where as during image acquisition there will be slight variations due to stage registration. | | False | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ```Chromatin``` ```DNA``` ```DNA + RNA``` ```Endogenous fluorophores``` ```Fluorochrome``` ```Lipid``` ```Metabolite``` ```Nucleic acid and protein``` ```Peptide``` ```Polysaccharide``` ```Protein``` ```RNA ```| True | -| acquisition_instrument_vendor | Allowable Value | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | ```Akoya Biosciences``` ```Andor``` ```BGI Genomics``` ```Bruker``` ```Cytiva``` ```Evident Scientific (Olympus)``` ```GE Healthcare``` ```Hamamatsu``` ```Huron Digital Pathology``` ```Illumina``` ```In-House``` ```Ionpath``` ```Keyence``` ```Leica Biosystems``` ```Leica Microsystems``` ```Motic``` ```NanoString``` ```Resolve Biosciences``` ```Sciex``` ```Standard BioTools (Fluidigm)``` ```Thermo Fisher Scientific``` ```Zeiss Microscopy``` | True | -| acquisition_instrument_model | Allowable Value | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```Aperio AT2``` ```Aperio CS2``` ```Axio Observer 3``` ```Axio Observer 5``` ```Axio Observer 7``` ```Axio Scan.Z1``` ```BZ-X710``` ```BZ-X800``` ```BZ-X810``` ```CosMx Spatial Molecular Imager``` ```Custom: Multiphoton``` ```Digital Spatial Profiler``` ```DM6 B``` ```DNBSEQ-T7``` ```EVOS M7000``` ```HiSeq 2500``` ```HiSeq 4000``` ```Hyperion Imaging System``` ```IN Cell Analyzer 2200``` ```Lightsheet 7``` ```MALDI timsTOF Flex Prototype``` ```MIBIscope``` ```MoticEasyScan One``` ```NanoZoomer 2.0-HT``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```NanoZoomer-SQ``` ```NextSeq 2000``` ```NextSeq 500``` ```NextSeq 550``` ```NovaSeq 6000``` ```NovaSeq X``` ```NovaSeq X Plus``` ```Orbitrap Eclipse Tribrid``` ```Orbitrap Fusion Lumos Tribrid``` ```Phenocycler-Fusion 1.0``` ```Phenocycler-Fusion 2.0``` ```PhenoImager Fusion``` ```Q Exactive``` ```Q Exactive HF``` ```Q Exactive UHMR``` ```QTRAP 5500``` ```Resolve Biosciences Molecular Cartography``` ```SCN400``` ```STELLARIS 5``` ```TissueScope LE Slide Scanner``` ```Unknown``` ```VS200 Slide Scanner``` ```Xenium Analyzer``` ```Zyla 4.2 sCMOS``` | True | -| source_storage_duration_unit | Allowable Value | The time duration unit of measurement | ```hour``` ```month``` ```day``` ```minute``` ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Allowable Value | The time unit of measurement |```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| tile_configuration | Allowable Value | This is how the tiles are configured for stitching. | ```Column-by-column``` ```Not applicable``` ```Row-by-row``` ```Snake-by-columns``` ```Snake-by-rows``` | False | -| scan_direction | Allowable Value | This is the direction of imaging, which is required for stitching. | ```Left-and-down``` ```Left-and-up``` ```Not applicable``` ```Right-and-down``` ```Right-and-up``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay ("Yes" or "No"). The CODEX analyte is protein. | ```Yes``` ```No``` | True | -| antibodies_path | Textfield | This is the location of the antibodies.tsv file relative to the root of the top level of the upload directory structure. This path should begin with "." and would likely be something like "./extras/antibodies.tsv". | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
diff --git a/docs/assays/metadata/testing/Visium-(no-probes).md b/docs/assays/metadata/Visium-(no-probes).md similarity index 100% rename from docs/assays/metadata/testing/Visium-(no-probes).md rename to docs/assays/metadata/Visium-(no-probes).md diff --git a/docs/assays/metadata/Visium-HD.md b/docs/assays/metadata/Visium-HD.md deleted file mode 100644 index c1cae516..00000000 --- a/docs/assays/metadata/Visium-HD.md +++ /dev/null @@ -1,33 +0,0 @@ ---- -layout: page ---- -# Visium-HD - -
Version 2 (current) - -## Version 2 (current) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | An internal field labs can use it to add whatever ID(s) they want or need for dataset validation and tracking. This could be a single ID (e.g., "Visium_9OLC_A4_S1") or a delimited list of IDs (e.g., “9OL; 9OLC.A2; Visium_9OLC_A4_S1”). This field will not be accessible to anyone outside of the consortium and no effort will be made to check if IDs provided by one data provider are also used by another. | | False | -| preparation_protocol_doi | Link | DOI for the protocols.io page that describes the assay or sample procurement and preparation. For example for an imaging assay, the protocol might begin with staining of a section and finalize with the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1. | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```DBiT-seq```, ```PhenoCycler```, ```CODEX```, ```Second Harmonic Generation (SHG)```, ```Seq-Scope``` | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx and Resolve, this is the area of the FOV (aka ROI) region being captured. For Xenium this is the total area of the FOV regions (aka ROI) being captured. For Stereo-Seq this is the number of beads. | | True | -| mapped_area_unit | Assigned Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2```, ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Assigned Value | The unit for spot size value. | ```um^2```, ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Assigned Value | Units corresponding to inter-spot distance | ```um``` | True | -| capture_area_id | Radio | Which capture area on the slide was used. For Visium this would be [A1, B1, C1, D1]. For HiFi this would be the lane on the flowcell. | ```A1```, ```B1```, ```C1```, ```D1```, ```Lane 1```, ```Lane 2```, ```Lane 3```, ```Lane 4```, ```Lane 5```, ```Lane 6```, ```Lane 7```, ```Lane 8``` | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Assigned Value | The unit for the permeabilization time. | ```minute``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| preparation_instrument_vendor | Assigned Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```Thermo Fisher Scientific```, ```SunChrom```, ```Leica Biosystems```, ```Roche Diagnostics```, ```In-House```, ```Not applicable```, ```Hamamatsu```, ```HTX Technologies```, ```10x Genomics``` | True | -| preparation_instrument_model | Assigned Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL```, ```ST5020 Multistainer```, ```Visium CytAssist```, ```SunCollect Sprayer```, ```Chromium X```, ```Chromium iX```, ```EVOS M7000```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```Discovery Ultra```, ```Sublimator```, ```Not applicable```, ```TM-Sprayer```, ```M5 Sprayer```, ```M3+ Sprayer```, ```Chromium Controller```, ```Chromium Connect``` | True | -| non_global_files | Textfield | A semicolon separated list of non-shared files to be included in the dataset. The path assumes the files are located in the "TOP/non-global/" directory. For example, for the file is TOP/non-global/lab_processed/images/1-tissue-boundary.geojson the value of this field would be "./lab_processed/images/1-tissue-boundary.geojson". After ingest, these files will be copied to the appropriate locations within the respective dataset directory tree. This field is used for internal HuBMAP processing. Examples for GeoMx and PhenoCycler are provided in the File Locations documentation: https://docs.google.com/document/d/1n2McSs9geA9Eli4QWQaB3c9R3wo5d5U1Xd57DWQfN5Q/edit#heading=h.1u82i4axggee | | False | - -
\ No newline at end of file diff --git a/docs/assays/metadata/VisiumNoProbes.md b/docs/assays/metadata/VisiumNoProbes.md deleted file mode 100644 index 2d741c42..00000000 --- a/docs/assays/metadata/VisiumNoProbes.md +++ /dev/null @@ -1,58 +0,0 @@ ---- -layout: page ---- -# Visium-(no-probes) - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version3 (current) - -## Version 3 - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Allowable Value | The unit for spot size value. | ```um^2``` ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Allowable Value | Units corresponding to inter-spot distance | ```um``` | True | -| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be ```A1, B1, C1, D1```. For HiFi this would be the lane on the flowcell. | ```A1``` ```B1``` ```C1``` ```D1``` ```Lane 1``` ```Lane 2``` ```Lane 3``` ```Lane 4``` ```Lane 5``` ```Lane 6``` ```Lane 7``` ```Lane 8``` | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | False | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| False | - -
- -
Version 2 - -## Version 2 - -| Attribute | Type | Description | Allowable Value | Required | -|-----------------------------|----------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------|------------| -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| dataset_type | Textfield | The specific type of dataset being produced. | | True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Textfield | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Textfield | The unit for spot size value. | | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Textfield | Units corresponding to inter-spot distance | | True | -| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be [A1, B1, C1, D1]. For HiFi this would be the lane on the flowcell. | [A1, B1, C1, D1, Lane 1, Lane 2, Lane 3, Lane 4, Lane 5, Lane 6, Lane 7, Lane 8] | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Textfield | The unit for the permeabilization time. | | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/VisiumWithProbes.md b/docs/assays/metadata/VisiumWithProbes.md deleted file mode 100644 index 89c07f31..00000000 --- a/docs/assays/metadata/VisiumWithProbes.md +++ /dev/null @@ -1,30 +0,0 @@ ---- -layout: page ---- -# Visium-(with-probes) -
Version 2 (current) - -## Version 2 (current) - -| Attribute | Type | Description | Allowable Values | Required | -|-------------------------------|-----------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------|------------| -| preparation_protocol_doi | Textfield | DOI for the protocols.io page that describes the assay or sample procurment and preparation. For example for an imaging assay, the protocol might include staining of a section through the creation of an OME-TIFF file. In this case the protocol would include any image processing steps required to create the OME-TIFF file. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| dataset_type | Allowable Value | The specific type of dataset being produced. | ```10X Multiome``` ```2D Imaging Mass Cytometry``` ```ATACseq``` ```Auto-fluorescence``` ```Cell DIVE``` ```CODEX``` ```Confocal``` ```CosMx``` ```CyCIF``` ```DBiT``` ```DESI``` ```Enhanced Stimulated Raman Spectroscopy (SRS)``` ```GeoMx (nCounter)``` ```GeoMx (NGS)``` ```HiFi-Slide``` ```Histology``` ```LC-MS``` ```Light Sheet``` ```MALDI``` ```MERFISH``` ```MIBI``` ```Molecular Cartography``` ```MUSIC``` ```nanoSPLITS``` ```PhenoCycler``` ```Resolve``` ```RNAseq``` ```RNAseq (with probes)``` ```Second Harmonic Generation (SHG)``` ```SIMS``` ```SNARE-seq2``` ```Stereo-seq``` ```Thick section Multiphoton MxIF``` ```Visium (no probes)``` ```Visium (with probes)``` ```Xenium```| True | -| contributors_path | Textfield | The path to the file with the ORCID IDs for all contributors of this dataset (e.g., "./extras/contributors.tsv" or "./contributors.tsv"). This is an internal metadata field that is just used for ingest. | | True | -| data_path | Textfield | The top level directory containing the raw and/or processed data. For a single dataset upload this might be "." where as for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For instance, if the data is within a directory called "TEST001-RK" use syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2" in which "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field that is just used for ingest. | | True | -| mapped_area_value | Numeric | For Visium, this is the area of spots that was covered by tissue within the captured area, not the total possible captured area which is fixed. For GeoMx this would be the area of the AOI being captured. For HiFi this is the summed area of the ROIs in a single flowcell lane. For CosMx, Xenium and Resolve, this is the area of the FOV (aka ROI) region being captured. | | True | -| mapped_area_unit | Allowable Value | The unit of measurement for the mapping area. For Visium and GeoMx this is typically um^2. | ```um^2``` ```mm^2``` | True | -| spot_size_value | Numeric | For assays where spots are used to define discrete capture areas, this is the area of a spot. | | True | -| spot_size_unit | Allowable Value | The unit for spot size value. | ```um^2``` ```mm^2``` | True | -| number_of_spots | Numeric | Number of capture spots within the mapped area. For Visium this would be the number of spots covered by tissue, while it's the number of spots within ROIs for HiFi. | | True | -| spot_spacing_value | Numeric | Approximate center-to-center distance between capture spots. Synonyms: Inter-Spot distance, Spot resolution, Pit size | | True | -| spot_spacing_unit | Allowable Value | Units corresponding to inter-spot distance | ```um``` | True | -| capture_area_id | Allowable Value | Which capture area on the slide was used. For Visium this would be ```A1, B1, C1, D1```. For HiFi this would be the lane on the flowcell. | ```A1``` ```B1``` ```C1``` ```D1``` ```Lane 1``` ```Lane 2``` ```Lane 3``` ```Lane 4``` ```Lane 5``` ```Lane 6``` ```Lane 7``` ```Lane 8``` | True | -| permeabilization_time_value | Numeric | Permeabilization time used for this tissue section. | | False | -| permeabilization_time_unit | Allowable Value | The unit for the permeabilization time. | ```minute``` | False | -| metadata_schema_id | Textfield | The string that serves as the definitive identifier for the metadata schema version and is readily interpretable by computers for data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| parent_sample_id | Textfield | Unique HuBMAP or SenNet identifier of the sample (i.e., block, section or suspension) used to perform this assay. For example, for a RNAseq assay, the parent would be the suspension, whereas, for one of the imaging assays, the parent would be the tissue section. If an assay comes from multiple parent samples then this should be a comma separated list. Example: HBM386.ZGKG.235, HBM672.MKPK.442 or SNT232.UBHJ.322, SNT329.ALSK.102 | | True | -| preparation_instrument_vendor | Allowable Value | The manufacturer of the instrument used to prepare (staining/processing) the sample for the assay. If an automatic slide staining method was indicated this field should list the manufacturer of the instrument. | ```10x Genomics``` ```Hamamatsu``` ```HTX Technologies``` ```In-House``` ```Leica Biosystems``` ```Not applicable``` ```Roche Diagnostics``` ```SunChrom``` ```Thermo Fisher Scientific``` | True | -| preparation_instrument_model | Allowable Value | Manufacturers of a staining system instrument may offer various versions (models) of that instrument with different features. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | ```AutoStainer XL``` ```Chromium Connect``` ```Chromium Controller``` ```Chromium iX``` ```Chromium X``` ```Discovery Ultra``` ```EVOS M7000``` ```M3+ Sprayer``` ```M5 Sprayer``` ```NanoZoomer S210``` ```NanoZoomer S360``` ```NanoZoomer S60``` ```Not applicable``` ```ST5020 Multistainer``` ```Sublimator``` ```SunCollect Sprayer``` ```TM-Sprayer``` ```Visium CytAssist ```| True | - -
diff --git a/docs/assays/metadata/WGS.md b/docs/assays/metadata/WGS.md deleted file mode 100644 index 20a9ee8b..00000000 --- a/docs/assays/metadata/WGS.md +++ /dev/null @@ -1,86 +0,0 @@ ---- -layout: page ---- -# WGS - -NOTE: Several versions of this metadata schema have been created over time. The (Latest) version contains most attributes, but there may be some deprecated attributes in the older versions for which data has been collected. HuBMAP is in the process of creating a reference which combines all of these versions into a single view. That reference will be available here once completed. - -
Version 1 (no longer accepting data) - -## Version 1 (no longer accepting data) - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| version | Allowable Value | Version of the schema to use when validating this metadata. | ['1'] | True | -| description | Textfield | Free-text description of this assay. | | True | -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['WGS'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| gdna_fragmentation_quality_assurance | Allowable Value | Is the gDNA integrity good enough for WGS? This is usually checked through running a gel. | ['Pass', 'Fail'] | True | -| dna_assay_input_value | Numeric | Amount of DNA input into library preparation | | True | -| dna_assay_input_unit | Allowable Value | Units of DNA input into library preparation | ['ug'] | False | -| library_construction_method | Textfield | Describes DNA library preparation kit. Modality of isolating gDNA, Fragmentation and generating sequencing libraries. | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | The adapter sequence to be used for adapter trimming starting with the 5' end. (eg. 5-ATCCTGAGAA) | | True | -| library_final_yield | Numeric | Total amount of library after final pcr amplification step | | True | -| library_final_yield_unit | Allowable Value | Total units of library after final pcr amplification step | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
- -
Version 0 - -## Version 0 - -| Attribute | Type | Description | Allowable Values | Required | -|---------------------------------------|-----------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------|------------| -| donor_id | Textfield | HuBMAP Display ID of the donor of the assayed tissue. | | True | -| tissue_id | Textfield | HuBMAP Display ID of the assayed tissue. | | True | -| execution_datetime | Datetime | Start date and time of assay, typically a date-time stamped folder generated by the acquisition instrument. YYYY-MM-DD hh:mm, where YYYY is the year, MM is the month with leading 0s, and DD is the day with leading 0s, hh is the hour with leading zeros, mm are the minutes with leading zeros. | | True | -| protocols_io_doi | Textfield | DOI for protocols.io referring to the protocol for this assay. | | True | -| operator | Textfield | Name of the person responsible for executing the assay. | | True | -| operator_email | Textfield | Email address for the operator. | | True | -| pi | Textfield | Name of the principal investigator responsible for the data. | | True | -| pi_email | Textfield | Email address for the principal investigator. | | True | -| assay_category | Allowable Value | Each assay is placed into one of the following 4 general categories: generation of images of microscopic entities, identification & quantitation of molecules by mass spectrometry, imaging mass spectrometry, and determination of nucleotide sequence. | ['sequence'] | True | -| assay_type | Allowable Value | The specific type of assay being executed. | ['WGS'] | True | -| analyte_class | Allowable Value | Analytes are the target molecules being measured with the assay. | ['DNA'] | True | -| is_targeted | Allowable Value | Specifies whether or not a specific molecule(s) is/are targeted for detection/measurement by the assay. | ['Yes','No'] | True | -| acquisition_instrument_vendor | Textfield | An acquisition instrument is the device that contains the signal detection hardware and signal processing software. Assays generate signals such as light of various intensities or color or signals representing the molecular mass. | | True | -| acquisition_instrument_model | Textfield | Manufacturers of an acquisition instrument may offer various versions (models) of that instrument with different features or sensitivities. Differences in features or sensitivities may be relevant to processing or interpretation of the data. | | True | -| gdna_fragmentation_quality_assurance | Allowable Value | Is the gDNA integrity good enough for WGS? This is usually checked through running a gel. | ['Pass', 'Fail'] | True | -| dna_assay_input_value | Numeric | Amount of DNA input into library preparation | | True | -| dna_assay_input_unit | Allowable Value | Units of DNA input into library preparation | ['ug'] | False | -| library_construction_method | Textfield | Describes DNA library preparation kit. Modality of isolating gDNA, Fragmentation and generating sequencing libraries. | | True | -| library_construction_protocols_io_doi | Textfield | A link to the protocol document containing the library construction method (including version) that was used. | | True | -| library_layout | Allowable Value | State whether the library was generated for single-end or paired end sequencing. | ['single-end', 'paired-end'] | True | -| library_adapter_sequence | Textfield | The adapter sequence to be used for adapter trimming starting with the 5' end. (eg. 5-ATCCTGAGAA) | | True | -| library_final_yield | Numeric | Total amount of library after final pcr amplification step | | True | -| library_final_yield_unit | Allowable Value | Total units of library after final pcr amplification step | ['ng'] | False | -| library_average_fragment_size | Numeric | Average size in basepairs (bp) of sequencing library fragments estimated via gel electrophoresis or bioanalyzer/tapestation. | | True | -| sequencing_reagent_kit | Textfield | Reagent kit used for sequencing | | True | -| sequencing_read_format | Textfield | Slash-delimited list of the number of sequencing cycles for, for example, Read1, i7 index, i5 index, and Read2. | | True | -| sequencing_read_percent_q30 | Numeric | Q30 is the weighted average of all the reads (e.g. # bases UMI * q30 UMI + # bases R2 * q30 R2 + ...) | | True | -| sequencing_phix_percent | Numeric | Percent PhiX loaded to the run | | True | -| contributors_path | Textfield | Relative path to file with ORCID IDs for contributors for this dataset. | | True | -| data_path | Textfield | Relative path to file or directory with instrument data. Downstream processing will depend on filename extension conventions. | | True | - -
diff --git a/docs/assays/metadata/testing/comet.md b/docs/assays/metadata/comet.md similarity index 100% rename from docs/assays/metadata/testing/comet.md rename to docs/assays/metadata/comet.md diff --git a/docs/assays/metadata/testing/cosmx-proteomics.md b/docs/assays/metadata/cosmx-proteomics.md similarity index 100% rename from docs/assays/metadata/testing/cosmx-proteomics.md rename to docs/assays/metadata/cosmx-proteomics.md diff --git a/docs/assays/metadata/testing/cosmx-transcriptomics.md b/docs/assays/metadata/cosmx-transcriptomics.md similarity index 100% rename from docs/assays/metadata/testing/cosmx-transcriptomics.md rename to docs/assays/metadata/cosmx-transcriptomics.md diff --git a/docs/assays/metadata/testing/cycif.md b/docs/assays/metadata/cycif.md similarity index 100% rename from docs/assays/metadata/testing/cycif.md rename to docs/assays/metadata/cycif.md diff --git a/docs/assays/metadata/testing/cytof.md b/docs/assays/metadata/cytof.md similarity index 100% rename from docs/assays/metadata/testing/cytof.md rename to docs/assays/metadata/cytof.md diff --git a/docs/assays/metadata/testing/dna-methylation.md b/docs/assays/metadata/dna-methylation.md similarity index 100% rename from docs/assays/metadata/testing/dna-methylation.md rename to docs/assays/metadata/dna-methylation.md diff --git a/docs/assays/metadata/testing/enhancedsrs.md b/docs/assays/metadata/enhancedsrs.md similarity index 100% rename from docs/assays/metadata/testing/enhancedsrs.md rename to docs/assays/metadata/enhancedsrs.md diff --git a/docs/assays/metadata/testing/facs.md b/docs/assays/metadata/facs.md similarity index 100% rename from docs/assays/metadata/testing/facs.md rename to docs/assays/metadata/facs.md diff --git a/docs/assays/metadata/testing/geomx.md b/docs/assays/metadata/geomx.md similarity index 100% rename from docs/assays/metadata/testing/geomx.md rename to docs/assays/metadata/geomx.md diff --git a/docs/assays/metadata/testing/hifi.md b/docs/assays/metadata/hifi.md similarity index 100% rename from docs/assays/metadata/testing/hifi.md rename to docs/assays/metadata/hifi.md diff --git a/docs/assays/metadata/iCLAP.md b/docs/assays/metadata/iCLAP.md deleted file mode 100644 index ce13e526..00000000 --- a/docs/assays/metadata/iCLAP.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -layout: page ---- -# iCLAP - -
Version 2.0 (use this one) - -## Version 2.0 (use this one) - -| Attribute Name | Type | Description | Allowable Values | Required | -|---------------|------|-------------|------------------|----------| -| lab_id | Textfield | A locally assigned identifier provided by the data provider for the dataset. It is used to reference an external metadata record that may be maintained independently, enabling traceability and supporting provenance tracking. Example: Visium_9OLC_A4_S1 | | False | -| source_storage_duration_value | Numeric | The length of time the sample was stored prior to processing it. For assays performed on tissue sections, this refers to how long the tissue section (e.g., slide) was stored before the assay began (e.g., imaging). For assays performed on suspensions, such as sequencing, it refers to how long the suspension was stored before library construction started. Example: 12 | | True | -| time_since_acquisition_instrument_calibration_value | Numeric | The length of time since the acquisition instrument was last serviced or calibrated. This provides a metric for assessing drift in data capture. Example: 10 | | False | -| contributors_path | Textfield | The name of the file containing the ORCID IDs for all contributors to this dataset. Example: ./contributors.csv | | True | -| data_path | Textfield | The top-level directory containing the raw and/or processed data. For a single dataset upload, this might be represented as ".", whereas for a data upload containing multiple datasets, this would be the directory name for the respective dataset. For example, if the data is within a directory named "TEST001-RK", use the syntax "./TEST001-RK" for this field. If there are multiple directory levels, use the format "./TEST001-RK/Run1/Pass2", where "Pass2" is the subdirectory where the single dataset's data is stored. This is an internal metadata field used solely for data ingestion. Example: ./TEST001-RK | | True | -| number_of_antibodies | Numeric | The number of antibodies used in the assay. If no antibodies were utilized, enter 0. Example: 5 | | True | -| number_of_biomarker_imaging_rounds | Numeric | The number of imaging rounds required to capture the tagged biomarkers. For CODEX, a biomarker imaging round includes steps such as (1) oligo application, (2) fluor application, and (3) washes. For Cell DIVE, it involves (1) the staining of a biomarker via secondary detection or direct conjugate, followed by (2) dye inactivation. Example: 3 | | True | -| number_of_total_imaging_rounds | Numeric | The total number of imaging rounds performed using a microscope to collect either autofluorescence/background or stained signals, such as those used in histological analysis. Example: 5 | | True | -| slide_id | Textfield | The unique identifier assigned to each slide, enabling users to determine which tissue sections were processed together on the same slide. It is recommended that data providers prefix the ID with the center name to prevent overlapping values across different centers. Example: VAN0071-PA-1-1_AF | | True | -| dataset_type | Assigned Value | The specific type of dataset being produced. Example: RNAseq | ```Visium HD```, ```4i```, ```LC-MS```, ```Thick section Multiphoton MxIF```, ```Light Sheet```, ```ATACseq```, ```Resolve```, ```HiFi-Slide```, ```COMET```, ```MPLEx```, ```10X Multiome```, ```MALDI```, ```Raman Imaging```, ```Histology```, ```Cell DIVE```, ```FACS```, ```MS Lipidomics```, ```Visium (no probes)```, ```MUSIC```, ```RNAseq```, ```GeoMx (NGS)```, ```GeoMx (nCounter)```, ```RNAseq (with probes)```, ```Singular Genomics G4X```, ```Molecular Cartography```, ```CosMx Transcriptomics```, ```MERFISH```, ```Pixel-seqV2```, ```2D Imaging Mass Cytometry```, ```Confocal```, ```seqFISH```, ```DART-FISH```, ```MIBI```, ```Olink```, ```Enhanced Stimulated Raman Spectroscopy (SRS)```, ```DESI```, ```Xenium```, ```iCLAP```, ```CyCIF```, ```SNARE-seq2```, ```nanoSPLITS```, ```STARmap```, ```Stereo-seq```, ```Visium (with probes)```, ```SIMS```, ```Auto-fluorescence```, ```CyTOF```, ```CosMx Proteomics```, ```Virtual Histology```, ```DBiT-seq``` | True | -| analyte_class | Assigned Value | The analyte class which is the target molecule that the assay is measuring. Example: DNA | ```Nucleic acid + protein```, ```Lipid + metabolite```, ```Collagen```, ```RNA```, ```Fluorochrome```, ```DNA```, ```Metabolite```, ```DNA + RNA```, ```Saturated lipid```, ```Lipid```, ```Lipid + metabolite + protein```, ```RNA + protein```, ```Peptide```, ```Protein```, ```Unsaturated lipid```, ```Endogenous fluorophore```, ```Chromatin```, ```Polysaccharide``` | True | -| acquisition_instrument_vendor | Assigned Value | The company that manufactures or supplies the acquisition instrument. An acquisition instrument is a device equipped with signal detection hardware and signal processing software. It captures signals produced by assays, such as variations in light intensity or color, or signals corresponding to molecular mass. If the instrument was custom-built or developed internally, enter "In-House". Example: Illumina | ```Complete Genomics```, ```Cytek Biosciences```, ```Thermo Fisher Scientific```, ```Sciex```, ```Vizgen```, ```Leica Microsystems```, ```Akoya Biosciences```, ```Keyence```, ```Andor```, ```Standard BioTools (Fluidigm)```, ```Leica Biosystems```, ```Zeiss Microscopy```, ```Ionpath```, ```Motic```, ```In-House```, ```Revvity```, ```Evident Scientific (Olympus)```, ```GE Healthcare```, ```Element Biosciences```, ```Hamamatsu```, ```Waters```, ```Bruker```, ```Illumina```, ```3DHISTECH```, ```Singular Genomics```, ```Huron Digital Pathology```, ```Resolve Biosciences```, ```NanoString```, ```Cytiva```, ```10x Genomics```, ```Microscopes International```, ```BGI Genomics``` | True | -| acquisition_instrument_model | Assigned Value | The specific model of the acquisition instrument, as manufacturers often offer various versions with differing features or sensitivities. These differences may be relevant to the processing or interpretation of the data. If the instrument was custom-built or developed internally, enter "In-House". If the model is unknown, enter "Unknown". Example: HiSeq 4000 | ```NovaSeq X```, ```NovaSeq X Plus```, ```Cytek Northern Lights```, ```Lightsheet 7```, ```Resolve Biosciences Molecular Cartography```, ```timsTOF HT```, ```timsTOF Pro 2```, ```timsTOF Pro```, ```timsTOF Ultra```, ```timsTOF Ultra 2```, ```timsTOF SCP```, ```Axio Scan.Z1```, ```MALDI timsTOF Flex Prototype```, ```CosMx Spatial Molecular Imager```, ```Unknown```, ```MERSCOPE Ultra```, ```Juno System```, ```timsTOF FleX```, ```Custom: Multiphoton```, ```CyTOF XT```, ```Helios```, ```EVOS M7000```, ```Aperio AT2```, ```Phenocycler-Fusion 2.0```, ```Axio Observer 5```, ```Axio Observer 7```, ```Axio Observer 3```, ```NanoZoomer-SQ```, ```NanoZoomer S210```, ```NanoZoomer S60```, ```NanoZoomer S360```, ```DM6 B```, ```MoticEasyScan One```, ```In-House```, ```NextSeq 500```, ```BZ-X710```, ```QTRAP 5500```, ```DMi8```, ```NextSeq 550```, ```HiSeq 2500```, ```HiSeq 4000```, ```NovaSeq 6000```, ```Opera Phenix Plus HCS```, ```SYNAPT G2-Si```, ```Q Exactive HF```, ```Orbitrap Fusion Tribrid```, ```Orbitrap Fusion Lumos Tribrid```, ```Q Exactive```, ```VS200 Slide Scanner```, ```Not applicable``` | True | -| source_storage_duration_unit | Assigned Value | The unit of measurement used to specify the source storage duration value. Example: hour | ```hour```, ```month```, ```day```, ```minute```, ```year``` | True | -| time_since_acquisition_instrument_calibration_unit | Assigned Value | The unit of measurement used to specify the time since acquisition instrument calibration value. Example: month | ```month```, ```day```, ```year``` | False | -| metadata_schema_id | Textfield | The unique string identifier for the metadata specification version, which is easily interpretable by computers for purposes of data validation and processing. Example: 22bc762a-5020-419d-b170-24253ed9e8d9 | | True | -| preparation_protocol_doi | Link | The DOI for the protocols.io page that details the assay or the procedures used for sample procurement and preparation. For example, in the case of an imaging assay, the protocol may start with tissue section staining and end with the generation of an OME-TIFF file. The documented protocol should also include any image processing steps involved in producing the final OME-TIFF. Example: https://dx.doi.org/10.17504/protocols.io.eq2lyno9qvx9/v1 | | True | -| is_targeted | Radio | Indicates whether a specific molecule or set of molecules is targeted for detection or measurement by the assay. Example: Yes | ```Yes```, ```No``` | True | -| antibodies_path | Textfield | The path to the antibodies.tsv file relative to the root directory of the upload structure. This path should start with "." and is typically formatted as "./extras/antibodies.tsv". Example: ./extras/antibodies.tsv | | True | -| parent_sample_id | Textfield | The unique identifier from HuBMAP or SenNet for the sample (such as a block, section, or suspension) used to perform the assay. For instance, in an RNAseq assay, the parent sample would be the suspension, while in imaging assays, it would be the tissue section. If the assay is derived from multiple parent samples, this field should contain a comma-separated list of identifiers. Example: HBM386.ZGKG.235, HBM672.MKPK.442 | | True | -| number_of_channels | Numeric | The number of fluorescent channels that are imaged during each cycle. Example: 3 | | True | - -
\ No newline at end of file diff --git a/docs/assays/metadata/testing/iclap.md b/docs/assays/metadata/iclap.md similarity index 100% rename from docs/assays/metadata/testing/iclap.md rename to docs/assays/metadata/iclap.md diff --git a/docs/assays/metadata/testing/illumina-spatial.md b/docs/assays/metadata/illumina-spatial.md similarity index 100% rename from docs/assays/metadata/testing/illumina-spatial.md rename to docs/assays/metadata/illumina-spatial.md diff --git a/docs/assays/metadata/testing/imc.md b/docs/assays/metadata/imc.md similarity index 100% rename from docs/assays/metadata/testing/imc.md rename to docs/assays/metadata/imc.md diff --git a/docs/assays/metadata/index.md b/docs/assays/metadata/index.md index ddda0ecb..efa18772 100644 --- a/docs/assays/metadata/index.md +++ b/docs/assays/metadata/index.md @@ -3,8 +3,7 @@ layout: page --- ## HuBMAP Metadata by Dataset Type -A list of available dataset types (data types from multiple supported assays), with a link [](EnhancedSRS "Attribute description") to the valid metadata attributes for each dataset type. The linked assay metadata pages list all attributes, as they have occurred, across any versions of the metadata specification for the given dataset type with the most current, valid set of attributes listed first on the page. The directory schema for each dataset type is also linked in the description column. - +A list of available dataset types (data types from multiple supported assays), with a link [](EnhancedSRS "Attribute description") to the valid metadata attributes for each dataset type. The linked assay metadata pages list all attributes, as they have occurred, across any versions of the metadata specification for the given dataset type with the most current, valid set of attributes listed first on the page. The directory schema for each dataset type is also linked in the description column. | Dataset Type | Description | |--------------|-------------| @@ -26,24 +25,24 @@ A list of available dataset types (data types from multiple supported assays), w | [IMC](https://docs.hubmapconsortium.org/assays/imc) [](IMC "Attribute description")| Combines standard immunohistochemistry with CyTOF mass cytometry to resolve the cellular localization of up to 40 proteins in a tissue sample. Link to [IMC directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/imc-2d/current/). | | [LC-MS](https://docs.hubmapconsortium.org/assays/lcms) [](LC-MS "Attribute description")| Coupling of liquid chromatography (LC) to mass spectrometry (MS). Link to [LC-MS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/lcms/current/). | | [Light Sheet](https://en.wikipedia.org/wiki/Light_sheet_fluorescence_microscopy) [](LightSheet "Attribute description")| A fluorescence imaging technique that uses a thin sheet of laser light to illuminate a sample, allowing for high-resolution, 3D imaging with reduced photobleaching and phototoxicity; particularly useful for imaging large, thick, or delicate biological samples, like developing embryos or organoids. Link to [Light Sheet directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/lightsheet/current/). | -| [MALDI-IMS](https://docs.hubmapconsortium.org/assays/maldi-ims) [](MALDI "Attribute description") | Matrix-assisted laser desorption/ionization (MALDI) imaging mass spectrometry (IMS) combines the sensitivity and molecular specificity of MS with the spatial fidelity of classical microscopy. Link to [MALDI-IMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/maldi/current/). | -| [MIBI](https://www.researchgate.net/figure/Multiplexed-ion-beam-imaging-workflow-for-high-resolution-spatial-proteomics-Here_fig1_349770840) [](MIBI "Attribute description") | Preserved tissue sections, mounted on conductive substrates are incubated with unique isotopic transition metal-tagged antibody reporters. An oxygen primary ion beam rasters the sample surface, ejecting and ionizing the isotope reporters. Their masses are subsequently measured via a mass analyzer. Link to [MIBI directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/mibi/current/). | -| [MERFISH](https://pubmed.ncbi.nlm.nih.gov/27241748/) [](MERFISH "Attribute description") | A spatial transcriptomics technology that allows for the simultaneous imaging of hundreds to thousands of RNA species within single cells, providing both copy number and spatial distribution information. Link to [MERFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/merfish/current/). | -| [MUSIC](https://www.nature.com/articles/s41586-024-07239-w) [](MUSIC "Attribute description") | A sequencing assay that allows profiling of gene expression, co-complexed DNA sequences, and RNA-chromatin interactions from the same single-cell nucleus. Both RNA and fragmented DNA within a nucleus are labelled with a unique cell barcode, enabling identification and matching of RNA and DNA sequences. Link to [MUSIC directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/music/current/). | -| [MxIF](https://pmc.ncbi.nlm.nih.gov/articles/PMC9959383/#) [](ThickSectionMultiphotonMxIF "Attribute description") | One version of MXIF (multiplexed fluorescence microscopy), an imaging platform whereby a large number of cellular and histological markers can be investigated on a single tissue section. Link to [TSM MxIF directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/thick-section-multiphoton-mxif/current/).| -| Pixel-seqV2 [](Pixel-seqV2 "Attribute description") | Pixel-seqV2 is a spatial transcriptomics method that utilizes polony gels to capture and sequence RNA, proteins or other molecules in tissues with high resolution. These polony gels are arrays of micron-scale DNA clusters, each containing a unique barcode, allowing for the mapping of molecules within their original spatial context in a tissue, thereby allowing researchers to study the spatial organization of cells and their gene expression profiles within tissues.| -| [RNAseq](https://docs.hubmapconsortium.org/assays/rnaseq) [](RNAseq "Attribute description") | While bulk RNAseq elucidates the average gene expression profile in cells comprising a tissue sample, single-cell RNAseq employs per-cell and per-molecule barcoding to enable single-cell resolution of the gene expression profile. Link to [RNAseq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq/current/).| -| [RNAseq with Probes](https://pmc.ncbi.nlm.nih.gov/articles/PMC5717752/#) [](RNAseqWithProbes "Attribute description") | Uses probes to capture and enrich specific regions of the RNA for targeted sequencing, allowing for in-depth analysis of those regions. Link to [RNAseq with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq-with-probes/current/).| -| Raman-Imaging [](Raman-Imaging "Attribute description") | Raman Imaging is a non-invasive technique that maps the unique chemical fingerprint of biological samples (cells, tissues) by capturing Raman scattering (light interacting with molecular vibrations) at each pixel, creating detailed molecular maps showing the distribution of proteins, lipids, DNA, and water. Link to [Raman Imaging directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/raman-imaging/current/). | -| [SHG](https://en.wikipedia.org/wiki/Second-harmonic_imaging_microscopy) [](SecondHarmonicGeneration "Attribute description") | Single-cycle Fluorescence Microscopy (SFM). A technique that utilizes the nonlinear optical phenomenon of SHG to image biological tissues and structures, particularly those containing collagen. Link to [SHG directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/second-harmonic-generation/current/).| -| [SeqFISH](https://docs.hubmapconsortium.org/assays/seqfish) [](seqFISH "Attribute description") | SeqFISH technology allows in situ imaging of multiple mRNAs using barcoding and fluorophore-labelled barcode readout-probes. Link to [SeqFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/seqfish/). _The consortium is no longer accepting data of this type_.| -| [SIMS](https://www.frontiersin.org/journals/chemistry/articles/10.3389/fchem.2023.1237408/full) [](SIMS "Attribute description") | Secondary-ion mass spectrometry (SIMS) is a technique used to analyze the composition of solid surfaces and thin films by sputtering the surface of the specimen. Link to [SIMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/sims/current/ "directory schema").| -| [Slide-seq](https://www.nature.com/articles/s41587-020-0739-1) [](Slide-seq "Attribute description") | Provides a scalable method for obtaining spatially resolved gene expression data at resolutions comparable to the sizes of individual cells. Link to [Slide-seq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/slide-seq/current/).| -| [SnareSeq2](https://www.nature.com/articles/s41596-021-00507-3) [](SnareSeq2 "Attribute description") | This method uses tagmentation within permeabilized and fixed single-nucleus isolates to capture accessible chromatin (AC) regions, followed by the capture and reverse transcription of RNA transcripts. Link to [SnareSeq2 directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/snareseq2/current/).| -| STARmap [](STARmap "Attribute description") | STARmap (Spatially-resolved Transcript Amplicon Readout Mapping) is a biomedical technology that enables the 3D mapping of gene expression within intact tissues at single-cell resolution. It combines hydrogel-tissue chemistry and in situ DNA sequencing to preserve a cell's location and identify which genes are active in that specific spatial context. [STARmap directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/starmap/current/). | -| [Visium No Probes](https://ostr.ccr.cancer.gov/emerging-technologies/spatial-biology/visium/) [](VisiumNoProbes "Attribute description") | A spatial transcriptomics solution that allows researchers to analyze gene expression patterns within the spatial context of a tissue. An in situ method that captures RNA transcripts within the tissue and then sequences them. Link to [Visium NP directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-no-probes/current/). | -| [Visium with Probes](https://ngisweden.scilifelab.se/methods/10x-genomics-visium-cytassist-for-ffpe-samples/) [](VisiumWithProbes "Attribute description") | Offers spatially resolved transcriptomics through the 10X Genomics Visium CytAssist, which combines histology with probe-based transcriptomics in a spatial context. Link to [Visium with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-with-probes/current/). | -| [WGS](https://en.wikipedia.org/wiki/Whole_genome_sequencing) [](WGS "Attribute description") | The process of determining the entire DNA sequence of an organism's genome at a single time. This entails sequencing all of an organism's chromosomal and mitochondrial DNA. Link to [WGS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/wgs/). _The consortium is no longer accepting data of this type_. | +| [MALDI-IMS](https://docs.hubmapconsortium.org/assays/maldi-ims) [](MALDI "Attribute description") | Matrix-assisted laser desorption/ionization (MALDI) imaging mass spectrometry (IMS) combines the sensitivity and molecular specificity of MS with the spatial fidelity of classical microscopy. Link to [MALDI-IMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/maldi/current/). | +| [MIBI](https://www.researchgate.net/figure/Multiplexed-ion-beam-imaging-workflow-for-high-resolution-spatial-proteomics-Here_fig1_349770840) [](MIBI "Attribute description") | Preserved tissue sections, mounted on conductive substrates are incubated with unique isotopic transition metal-tagged antibody reporters. An oxygen primary ion beam rasters the sample surface, ejecting and ionizing the isotope reporters. Their masses are subsequently measured via a mass analyzer. Link to [MIBI directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/mibi/current/). | +| [MERFISH](https://pubmed.ncbi.nlm.nih.gov/27241748/) [](MERFISH "Attribute description") | A spatial transcriptomics technology that allows for the simultaneous imaging of hundreds to thousands of RNA species within single cells, providing both copy number and spatial distribution information. Link to [MERFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/merfish/current/). | +| [MUSIC](https://www.nature.com/articles/s41586-024-07239-w) [](MUSIC "Attribute description") | A sequencing assay that allows profiling of gene expression, co-complexed DNA sequences, and RNA-chromatin interactions from the same single-cell nucleus. Both RNA and fragmented DNA within a nucleus are labelled with a unique cell barcode, enabling identification and matching of RNA and DNA sequences. Link to [MUSIC directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/music/current/). | +| [MxIF](https://pmc.ncbi.nlm.nih.gov/articles/PMC9959383/#) [](ThickSectionMultiphotonMxIF "Attribute description") | One version of MXIF (multiplexed fluorescence microscopy), an imaging platform whereby a large number of cellular and histological markers can be investigated on a single tissue section. Link to [TSM MxIF directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/thick-section-multiphoton-mxif/current/).| +| Pixel-seqV2 [](Pixel-seqV2 "Attribute description") | Pixel-seqV2 is a spatial transcriptomics method that utilizes polony gels to capture and sequence RNA, proteins or other molecules in tissues with high resolution. These polony gels are arrays of micron-scale DNA clusters, each containing a unique barcode, allowing for the mapping of molecules within their original spatial context in a tissue, thereby allowing researchers to study the spatial organization of cells and their gene expression profiles within tissues.| +| [RNAseq](https://docs.hubmapconsortium.org/assays/rnaseq) [](RNAseq "Attribute description") | While bulk RNAseq elucidates the average gene expression profile in cells comprising a tissue sample, single-cell RNAseq employs per-cell and per-molecule barcoding to enable single-cell resolution of the gene expression profile. Link to [RNAseq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq/current/).| +| [RNAseq with Probes](https://pmc.ncbi.nlm.nih.gov/articles/PMC5717752/#) [](RNAseqWithProbes "Attribute description") | Uses probes to capture and enrich specific regions of the RNA for targeted sequencing, allowing for in-depth analysis of those regions. Link to [RNAseq with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/rnaseq-with-probes/current/).| +| Raman-Imaging [](Raman-Imaging "Attribute description") | Raman Imaging is a non-invasive technique that maps the unique chemical fingerprint of biological samples (cells, tissues) by capturing Raman scattering (light interacting with molecular vibrations) at each pixel, creating detailed molecular maps showing the distribution of proteins, lipids, DNA, and water. Link to [Raman Imaging directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/raman-imaging/current/). | +| [SHG](https://en.wikipedia.org/wiki/Second-harmonic_imaging_microscopy) [](SecondHarmonicGeneration "Attribute description") | Single-cycle Fluorescence Microscopy (SFM). A technique that utilizes the nonlinear optical phenomenon of SHG to image biological tissues and structures, particularly those containing collagen. Link to [SHG directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/second-harmonic-generation/current/).| +| [SeqFISH](https://docs.hubmapconsortium.org/assays/seqfish) [](seqFISH "Attribute description") | SeqFISH technology allows in situ imaging of multiple mRNAs using barcoding and fluorophore-labelled barcode readout-probes. Link to [SeqFISH directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/seqfish/). _The consortium is no longer accepting data of this type_.| +| [SIMS](https://www.frontiersin.org/journals/chemistry/articles/10.3389/fchem.2023.1237408/full) [](SIMS "Attribute description") | Secondary-ion mass spectrometry (SIMS) is a technique used to analyze the composition of solid surfaces and thin films by sputtering the surface of the specimen. Link to [SIMS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/sims/current/ "directory schema").| +| [Slide-seq](https://www.nature.com/articles/s41587-020-0739-1) [](Slide-seq "Attribute description") | Provides a scalable method for obtaining spatially resolved gene expression data at resolutions comparable to the sizes of individual cells. Link to [Slide-seq directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/slide-seq/current/).| +| [SnareSeq2](https://www.nature.com/articles/s41596-021-00507-3) [](SnareSeq2 "Attribute description") | This method uses tagmentation within permeabilized and fixed single-nucleus isolates to capture accessible chromatin (AC) regions, followed by the capture and reverse transcription of RNA transcripts. Link to [SnareSeq2 directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/snareseq2/current/).| +| STARmap [](STARmap "Attribute description") | STARmap (Spatially-resolved Transcript Amplicon Readout Mapping) is a biomedical technology that enables the 3D mapping of gene expression within intact tissues at single-cell resolution. It combines hydrogel-tissue chemistry and in situ DNA sequencing to preserve a cell's location and identify which genes are active in that specific spatial context. [STARmap directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/starmap/current/). | +| [Visium No Probes](https://ostr.ccr.cancer.gov/emerging-technologies/spatial-biology/visium/) [](VisiumNoProbes "Attribute description") | A spatial transcriptomics solution that allows researchers to analyze gene expression patterns within the spatial context of a tissue. An in situ method that captures RNA transcripts within the tissue and then sequences them. Link to [Visium NP directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-no-probes/current/). | +| [Visium with Probes](https://ngisweden.scilifelab.se/methods/10x-genomics-visium-cytassist-for-ffpe-samples/) [](VisiumWithProbes "Attribute description") | Offers spatially resolved transcriptomics through the 10X Genomics Visium CytAssist, which combines histology with probe-based transcriptomics in a spatial context. Link to [Visium with Probes directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/visium-with-probes/current/). | +| [WGS](https://en.wikipedia.org/wiki/Whole_genome_sequencing) [](WGS "Attribute description") | The process of determining the entire DNA sequence of an organism's genome at a single time. This entails sequencing all of an organism's chromosomal and mitochondrial DNA. Link to [WGS directory schema](https://hubmapconsortium.github.io/ingest-validation-tools/wgs/). _The consortium is no longer accepting data of this type_. | diff --git a/docs/assays/metadata/testing/info3.png b/docs/assays/metadata/testing/info3.png deleted file mode 100644 index 811c300e2d8c4f154f3319a208424e3e135518ef..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 2038 zcmY*ae>~IqAOFl)Hu;exQgXVuJGafukC|U`JqzzH|GoDQrssKvxJFpL^@7w-dRcJ=Jl3W4c+D|&GG%ZyP<#~i80Dz9Z z+CZn5!4;|tIqZO7elW=!%ZQ6b(3o*_7D5=yQ%wT^ZjX>hV;p7iE$FN$HWzRG>Za7% zg3ZKR2RV>zNxc26Xtqa6Jj*}DCxDT1l;O;@-b2uZ;e=QfLM)3MRxqMV(bk+yb7J3F{)0-lh} z<7~wx1)mv5Bzsjg(`d3{RS0$-HrQNqhB{~40wd6^34T#?VvNR;REYPU? z_~P!;JKo;@Vll$x!_+`+MV*IRns0M?xlt&kaZ5PGw?ixFwQI$xlT%sxu$aJc*HiQk zU9G2nu{P!6Tyb7Bb*wTsw&ml(P2RCy@7twDsik{1qMta;K3oXgmm@k%6?HL$Id$sXa;d!ZP82~YyzFQprp0NEM%_Y>MiA}dx4 z=m`)9eg&kFrHt}z*GaN{3Ya3;^W6#+u)EZ8$hEU8-94478M-zPxv(KD@ifB{IWm~x zk`ZF4yzw#}v3xRgbpAu*FJ3ZDDVp?jXYP%U%Z5UOCF~A+H}mU3V5gg9F5CIXp^4y& zE()s%n@0Zy(?1~I5Ui+F#!+C_=K|}gy#xrZP|y5|;Rr+#R-Bd-I`U+lpb-XYdl(Sb z3@qALq0=7Cy=Xpa{GsmTxI80Abn)S6@m~Gc?)Gxwca>}Jhzt&3PA0E+R3nF;(G~iv zQPGSHjz}|a1_fypvte_T_Zj)JBlmlU&%I?McF1xHI3X11>N9WYz=5zrSiZgw#ZW7ej7ZIk4iEeBDg?)FjliKGBHf}9XeG;1`&;J5tHs2SX#j9Gj0HrOIB z-3cUME5S0N?x*J-R>(@?ZLvXe&VK(6?S9N-VmI=>wPsPfJgoL5M_~F$+<&U^o-*-V&lW9<{!+DifOh_R&^1+Df#7 zhw{D(St-^Fq$p_hd0dW}=mhD({uB;g@*5vLZ*hD3E-|2C$BdT*^wxCY(dre+?%STuFn%6#*8Z0Kp!Z*Br928mWk_wrqX=9fxFXdazr zg$5K|y@aOug;woV&OSdKyX6zBbc;pF_%rWN-Jiq}m1gkhmftW>&dF!oJPErNR&3D5 zTX!9I{5tl`<{D)0U7CiRk$LNs>~7E0{z9ZzN>0p!7f?)D_coDqf+8*-4AjdVYTLMk zFqDsfIW)yAJY%rD@6X3Y^1v+91?ai;n?1J@&wb+D_~Lqr{dS$jw>6iaT6HBU@7B6* zwhbcWFB~ZZNj)V)=W#<{;*hZW8-6w3{)5F55nsx@KiGF;DhNMs3B6 zbsnsn{7ALDo+fK52nS1sMBSaY$mNf%)5GW3U&L|$n353F*GFvMU_#sHAGA)aDvGLU z&dRFcq!o(o;OOLY28tKOv}8;kRs?j%755*V?3X;VE&rObeYZnGSzs$HyveP=&!{!~ z>O|iLZeu?#ScCGaFK(c4Tyxb>a{zYlu;+!ktep?jKmKlDuBrFnF1q*{6OqCwb-Y5W zsd|3?w6bwyo1O8&dvx5My+@S({hPWHhf_L-H5N5b823)MRoA=R8yIBN z@SN>~d&dj{m)gX0%Qd|G5@|#2$zvHGC_lC$1PyG3M%Phkv1v`E&1EeNSVD^Ram@J& zI~=c=kAF9U>r2VaCWOu$Yo{fiNopIK+1ODro!_^~bLJ*3Jm@x~=d}+l@_U!>&tE9* ztniR1OCsJE1xoyd^7tsX>D_r}soIqKBuL@dSqRjDc%XpikEZ<964^TQiIR9KMh+>; zSX?Wk%;5igKW%=swu=pCGQ2&xCN7EJx#g}`=&U?mP3jeLP3Eev{D^&_cJx)9tJkZX T8k8ku_2=v9=0mLC7m@ilud0$r diff --git a/docs/assays/metadata/testing/thicksectionmultiphotonmxif.md b/docs/assays/metadata/thicksectionmultiphotonmxif.md similarity index 100% rename from docs/assays/metadata/testing/thicksectionmultiphotonmxif.md rename to docs/assays/metadata/thicksectionmultiphotonmxif.md diff --git a/docs/assays/metadata/testing/visium-hd.md b/docs/assays/metadata/visium-hd.md similarity index 100% rename from docs/assays/metadata/testing/visium-hd.md rename to docs/assays/metadata/visium-hd.md diff --git a/docs/assays/metadata/testing/visiumwithprobes.md b/docs/assays/metadata/visiumwithprobes.md similarity index 100% rename from docs/assays/metadata/testing/visiumwithprobes.md rename to docs/assays/metadata/visiumwithprobes.md diff --git a/docs/assays/metadata/testing/wgs.md b/docs/assays/metadata/wgs.md similarity index 100% rename from docs/assays/metadata/testing/wgs.md rename to docs/assays/metadata/wgs.md