[
  {
    "name": "metadata-show-tables",
    "source": "metadata",
    "sql": "SHOW TABLES"
  },
  {
    "name": "metadata-describe-cancer_study",
    "source": "metadata",
    "sql": "DESCRIBE TABLE cancer_study"
  },
  {
    "name": "metadata-describe-genomic_event_derived",
    "source": "metadata",
    "sql": "DESCRIBE TABLE genomic_event_derived"
  },
  {
    "name": "metadata-describe-sample_to_gene_panel_derived",
    "source": "metadata",
    "sql": "DESCRIBE TABLE sample_to_gene_panel_derived"
  },
  {
    "name": "metadata-describe-clinical_data_derived",
    "source": "metadata",
    "sql": "DESCRIBE TABLE clinical_data_derived"
  },
  {
    "name": "metadata-describe-sample_derived",
    "source": "metadata",
    "sql": "DESCRIBE TABLE sample_derived"
  },
  {
    "name": "metadata-describe-gene",
    "source": "metadata",
    "sql": "DESCRIBE TABLE gene"
  },
  {
    "name": "metadata-describe-genetic_profile",
    "source": "metadata",
    "sql": "DESCRIBE TABLE genetic_profile"
  },
  {
    "name": "metadata-version-timezone",
    "source": "metadata",
    "sql": "SELECT version(), timezone()"
  },
  {
    "name": "metadata-settings",
    "source": "metadata",
    "sql": "SELECT name, value FROM system.settings LIMIT 1000"
  },
  {
    "name": "inventory",
    "source": "metadata",
    "sql": "SELECT name, total_rows, total_bytes FROM system.tables WHERE database=currentDatabase() ORDER BY name"
  },
  {
    "name": "lookup-estimate-gene",
    "source": "metadata",
    "sql": "SELECT hugo_gene_symbol, entrez_gene_id FROM gene WHERE hugo_gene_symbol='TP53'"
  },
  {
    "name": "lookup-estimate-study",
    "source": "metadata",
    "sql": "SELECT cancer_study_identifier FROM cancer_study WHERE cancer_study_identifier='msk_chord_2024'"
  },
  {
    "name": "160-before-TP53-mutation",
    "source": "pr160",
    "sql": "SELECT * FROM gene_alteration_frequency_by_cancer_type(preference='all_studies_non_redundant', gene='TP53', alteration='mutation')"
  },
  {
    "name": "160-after-TP53-mutation",
    "source": "pr160",
    "sql": "WITH cohort AS (\n    SELECT cancer_study_identifier\n    FROM cancer_study_query_preferences\n    WHERE preference_name = 'all_studies_non_redundant'\n),\nsample_cancer_type AS (\n    SELECT cd.sample_unique_id, cd.attribute_value AS cancer_type\n    FROM clinical_data_derived cd\n    JOIN cohort c USING (cancer_study_identifier)\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\naltered AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT ged.sample_unique_id) AS altered_samples\n    FROM genomic_event_derived ged\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    WHERE ged.hugo_gene_symbol = 'TP53'\n      AND ged.off_panel = 0\n      AND (\n        ('mutation' = 'mutation'\n            AND ged.variant_type = 'mutation'\n            AND ged.mutation_status != 'UNCALLED')\n        OR ('mutation' = 'amplification'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = 2)\n        OR ('mutation' = 'deep_deletion'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = -2)\n        OR ('mutation' = 'structural_variant'\n            AND ged.variant_type = 'structural_variant')\n      )\n    GROUP BY sct.cancer_type\n),\nprofiled_samples_for_gene AS (\n    -- Map the user-facing alteration token to the alteration_type stored\n    -- on sample_to_gene_panel_derived. Same gene-in-panel-or-WES branch\n    -- as the mutation view, but with the matching alteration_type filter.\n    SELECT stgp.sample_unique_id, stgp.cancer_study_identifier\n    FROM sample_to_gene_panel_derived stgp\n    JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    -- IN-subquery instead of JOIN gene: resolves the symbol to its entrez\n    -- id(s) once, rather than joining every panel row against the gene\n    -- table before filtering. COUNT(DISTINCT) downstream makes the result\n    -- identical even for symbols with several (or duplicated) gene rows.\n    -- IS NOT NULL keeps NULL keys unmatched (as the equality JOIN did) even\n    -- under transform_null_in=1.\n    WHERE gpl.gene_id IN (\n        SELECT entrez_gene_id FROM gene\n        WHERE hugo_gene_symbol = 'TP53' AND entrez_gene_id IS NOT NULL)\n      AND stgp.alteration_type = multiIf(\n          'mutation' = 'mutation',           'MUTATION_EXTENDED',\n          'mutation' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'mutation' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'mutation' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n    UNION ALL\n    SELECT sample_unique_id, cancer_study_identifier\n    FROM sample_to_gene_panel_derived\n    WHERE gene_panel_id = 'WES'\n      AND alteration_type = multiIf(\n          'mutation' = 'mutation',           'MUTATION_EXTENDED',\n          'mutation' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'mutation' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'mutation' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n),\nprofiled AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT p.sample_unique_id) AS profiled_samples\n    FROM profiled_samples_for_gene p\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    GROUP BY sct.cancer_type\n)\nSELECT a.cancer_type,\n       a.altered_samples,\n       p.profiled_samples,\n       ROUND(a.altered_samples * 100.0 / NULLIF(p.profiled_samples, 0), 1) AS frequency_pct\nFROM altered a\nJOIN profiled p USING (cancer_type)\nWHERE p.profiled_samples >= 50"
  },
  {
    "name": "160-before-TP53-amplification",
    "source": "pr160",
    "sql": "SELECT * FROM gene_alteration_frequency_by_cancer_type(preference='all_studies_non_redundant', gene='TP53', alteration='amplification')"
  },
  {
    "name": "160-after-TP53-amplification",
    "source": "pr160",
    "sql": "WITH cohort AS (\n    SELECT cancer_study_identifier\n    FROM cancer_study_query_preferences\n    WHERE preference_name = 'all_studies_non_redundant'\n),\nsample_cancer_type AS (\n    SELECT cd.sample_unique_id, cd.attribute_value AS cancer_type\n    FROM clinical_data_derived cd\n    JOIN cohort c USING (cancer_study_identifier)\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\naltered AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT ged.sample_unique_id) AS altered_samples\n    FROM genomic_event_derived ged\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    WHERE ged.hugo_gene_symbol = 'TP53'\n      AND ged.off_panel = 0\n      AND (\n        ('amplification' = 'mutation'\n            AND ged.variant_type = 'mutation'\n            AND ged.mutation_status != 'UNCALLED')\n        OR ('amplification' = 'amplification'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = 2)\n        OR ('amplification' = 'deep_deletion'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = -2)\n        OR ('amplification' = 'structural_variant'\n            AND ged.variant_type = 'structural_variant')\n      )\n    GROUP BY sct.cancer_type\n),\nprofiled_samples_for_gene AS (\n    -- Map the user-facing alteration token to the alteration_type stored\n    -- on sample_to_gene_panel_derived. Same gene-in-panel-or-WES branch\n    -- as the mutation view, but with the matching alteration_type filter.\n    SELECT stgp.sample_unique_id, stgp.cancer_study_identifier\n    FROM sample_to_gene_panel_derived stgp\n    JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    -- IN-subquery instead of JOIN gene: resolves the symbol to its entrez\n    -- id(s) once, rather than joining every panel row against the gene\n    -- table before filtering. COUNT(DISTINCT) downstream makes the result\n    -- identical even for symbols with several (or duplicated) gene rows.\n    -- IS NOT NULL keeps NULL keys unmatched (as the equality JOIN did) even\n    -- under transform_null_in=1.\n    WHERE gpl.gene_id IN (\n        SELECT entrez_gene_id FROM gene\n        WHERE hugo_gene_symbol = 'TP53' AND entrez_gene_id IS NOT NULL)\n      AND stgp.alteration_type = multiIf(\n          'amplification' = 'mutation',           'MUTATION_EXTENDED',\n          'amplification' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'amplification' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'amplification' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n    UNION ALL\n    SELECT sample_unique_id, cancer_study_identifier\n    FROM sample_to_gene_panel_derived\n    WHERE gene_panel_id = 'WES'\n      AND alteration_type = multiIf(\n          'amplification' = 'mutation',           'MUTATION_EXTENDED',\n          'amplification' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'amplification' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'amplification' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n),\nprofiled AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT p.sample_unique_id) AS profiled_samples\n    FROM profiled_samples_for_gene p\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    GROUP BY sct.cancer_type\n)\nSELECT a.cancer_type,\n       a.altered_samples,\n       p.profiled_samples,\n       ROUND(a.altered_samples * 100.0 / NULLIF(p.profiled_samples, 0), 1) AS frequency_pct\nFROM altered a\nJOIN profiled p USING (cancer_type)\nWHERE p.profiled_samples >= 50"
  },
  {
    "name": "160-before-TP53-deep_deletion",
    "source": "pr160",
    "sql": "SELECT * FROM gene_alteration_frequency_by_cancer_type(preference='all_studies_non_redundant', gene='TP53', alteration='deep_deletion')"
  },
  {
    "name": "160-after-TP53-deep_deletion",
    "source": "pr160",
    "sql": "WITH cohort AS (\n    SELECT cancer_study_identifier\n    FROM cancer_study_query_preferences\n    WHERE preference_name = 'all_studies_non_redundant'\n),\nsample_cancer_type AS (\n    SELECT cd.sample_unique_id, cd.attribute_value AS cancer_type\n    FROM clinical_data_derived cd\n    JOIN cohort c USING (cancer_study_identifier)\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\naltered AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT ged.sample_unique_id) AS altered_samples\n    FROM genomic_event_derived ged\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    WHERE ged.hugo_gene_symbol = 'TP53'\n      AND ged.off_panel = 0\n      AND (\n        ('deep_deletion' = 'mutation'\n            AND ged.variant_type = 'mutation'\n            AND ged.mutation_status != 'UNCALLED')\n        OR ('deep_deletion' = 'amplification'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = 2)\n        OR ('deep_deletion' = 'deep_deletion'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = -2)\n        OR ('deep_deletion' = 'structural_variant'\n            AND ged.variant_type = 'structural_variant')\n      )\n    GROUP BY sct.cancer_type\n),\nprofiled_samples_for_gene AS (\n    -- Map the user-facing alteration token to the alteration_type stored\n    -- on sample_to_gene_panel_derived. Same gene-in-panel-or-WES branch\n    -- as the mutation view, but with the matching alteration_type filter.\n    SELECT stgp.sample_unique_id, stgp.cancer_study_identifier\n    FROM sample_to_gene_panel_derived stgp\n    JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    -- IN-subquery instead of JOIN gene: resolves the symbol to its entrez\n    -- id(s) once, rather than joining every panel row against the gene\n    -- table before filtering. COUNT(DISTINCT) downstream makes the result\n    -- identical even for symbols with several (or duplicated) gene rows.\n    -- IS NOT NULL keeps NULL keys unmatched (as the equality JOIN did) even\n    -- under transform_null_in=1.\n    WHERE gpl.gene_id IN (\n        SELECT entrez_gene_id FROM gene\n        WHERE hugo_gene_symbol = 'TP53' AND entrez_gene_id IS NOT NULL)\n      AND stgp.alteration_type = multiIf(\n          'deep_deletion' = 'mutation',           'MUTATION_EXTENDED',\n          'deep_deletion' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'deep_deletion' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'deep_deletion' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n    UNION ALL\n    SELECT sample_unique_id, cancer_study_identifier\n    FROM sample_to_gene_panel_derived\n    WHERE gene_panel_id = 'WES'\n      AND alteration_type = multiIf(\n          'deep_deletion' = 'mutation',           'MUTATION_EXTENDED',\n          'deep_deletion' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'deep_deletion' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'deep_deletion' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n),\nprofiled AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT p.sample_unique_id) AS profiled_samples\n    FROM profiled_samples_for_gene p\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    GROUP BY sct.cancer_type\n)\nSELECT a.cancer_type,\n       a.altered_samples,\n       p.profiled_samples,\n       ROUND(a.altered_samples * 100.0 / NULLIF(p.profiled_samples, 0), 1) AS frequency_pct\nFROM altered a\nJOIN profiled p USING (cancer_type)\nWHERE p.profiled_samples >= 50"
  },
  {
    "name": "160-before-TP53-structural_variant",
    "source": "pr160",
    "sql": "SELECT * FROM gene_alteration_frequency_by_cancer_type(preference='all_studies_non_redundant', gene='TP53', alteration='structural_variant')"
  },
  {
    "name": "160-after-TP53-structural_variant",
    "source": "pr160",
    "sql": "WITH cohort AS (\n    SELECT cancer_study_identifier\n    FROM cancer_study_query_preferences\n    WHERE preference_name = 'all_studies_non_redundant'\n),\nsample_cancer_type AS (\n    SELECT cd.sample_unique_id, cd.attribute_value AS cancer_type\n    FROM clinical_data_derived cd\n    JOIN cohort c USING (cancer_study_identifier)\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\naltered AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT ged.sample_unique_id) AS altered_samples\n    FROM genomic_event_derived ged\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    WHERE ged.hugo_gene_symbol = 'TP53'\n      AND ged.off_panel = 0\n      AND (\n        ('structural_variant' = 'mutation'\n            AND ged.variant_type = 'mutation'\n            AND ged.mutation_status != 'UNCALLED')\n        OR ('structural_variant' = 'amplification'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = 2)\n        OR ('structural_variant' = 'deep_deletion'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = -2)\n        OR ('structural_variant' = 'structural_variant'\n            AND ged.variant_type = 'structural_variant')\n      )\n    GROUP BY sct.cancer_type\n),\nprofiled_samples_for_gene AS (\n    -- Map the user-facing alteration token to the alteration_type stored\n    -- on sample_to_gene_panel_derived. Same gene-in-panel-or-WES branch\n    -- as the mutation view, but with the matching alteration_type filter.\n    SELECT stgp.sample_unique_id, stgp.cancer_study_identifier\n    FROM sample_to_gene_panel_derived stgp\n    JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    -- IN-subquery instead of JOIN gene: resolves the symbol to its entrez\n    -- id(s) once, rather than joining every panel row against the gene\n    -- table before filtering. COUNT(DISTINCT) downstream makes the result\n    -- identical even for symbols with several (or duplicated) gene rows.\n    -- IS NOT NULL keeps NULL keys unmatched (as the equality JOIN did) even\n    -- under transform_null_in=1.\n    WHERE gpl.gene_id IN (\n        SELECT entrez_gene_id FROM gene\n        WHERE hugo_gene_symbol = 'TP53' AND entrez_gene_id IS NOT NULL)\n      AND stgp.alteration_type = multiIf(\n          'structural_variant' = 'mutation',           'MUTATION_EXTENDED',\n          'structural_variant' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'structural_variant' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'structural_variant' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n    UNION ALL\n    SELECT sample_unique_id, cancer_study_identifier\n    FROM sample_to_gene_panel_derived\n    WHERE gene_panel_id = 'WES'\n      AND alteration_type = multiIf(\n          'structural_variant' = 'mutation',           'MUTATION_EXTENDED',\n          'structural_variant' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'structural_variant' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'structural_variant' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n),\nprofiled AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT p.sample_unique_id) AS profiled_samples\n    FROM profiled_samples_for_gene p\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    GROUP BY sct.cancer_type\n)\nSELECT a.cancer_type,\n       a.altered_samples,\n       p.profiled_samples,\n       ROUND(a.altered_samples * 100.0 / NULLIF(p.profiled_samples, 0), 1) AS frequency_pct\nFROM altered a\nJOIN profiled p USING (cancer_type)\nWHERE p.profiled_samples >= 50"
  },
  {
    "name": "160-before-ERBB2-mutation",
    "source": "pr160",
    "sql": "SELECT * FROM gene_alteration_frequency_by_cancer_type(preference='all_studies_non_redundant', gene='ERBB2', alteration='mutation')"
  },
  {
    "name": "160-after-ERBB2-mutation",
    "source": "pr160",
    "sql": "WITH cohort AS (\n    SELECT cancer_study_identifier\n    FROM cancer_study_query_preferences\n    WHERE preference_name = 'all_studies_non_redundant'\n),\nsample_cancer_type AS (\n    SELECT cd.sample_unique_id, cd.attribute_value AS cancer_type\n    FROM clinical_data_derived cd\n    JOIN cohort c USING (cancer_study_identifier)\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\naltered AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT ged.sample_unique_id) AS altered_samples\n    FROM genomic_event_derived ged\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    WHERE ged.hugo_gene_symbol = 'ERBB2'\n      AND ged.off_panel = 0\n      AND (\n        ('mutation' = 'mutation'\n            AND ged.variant_type = 'mutation'\n            AND ged.mutation_status != 'UNCALLED')\n        OR ('mutation' = 'amplification'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = 2)\n        OR ('mutation' = 'deep_deletion'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = -2)\n        OR ('mutation' = 'structural_variant'\n            AND ged.variant_type = 'structural_variant')\n      )\n    GROUP BY sct.cancer_type\n),\nprofiled_samples_for_gene AS (\n    -- Map the user-facing alteration token to the alteration_type stored\n    -- on sample_to_gene_panel_derived. Same gene-in-panel-or-WES branch\n    -- as the mutation view, but with the matching alteration_type filter.\n    SELECT stgp.sample_unique_id, stgp.cancer_study_identifier\n    FROM sample_to_gene_panel_derived stgp\n    JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    -- IN-subquery instead of JOIN gene: resolves the symbol to its entrez\n    -- id(s) once, rather than joining every panel row against the gene\n    -- table before filtering. COUNT(DISTINCT) downstream makes the result\n    -- identical even for symbols with several (or duplicated) gene rows.\n    -- IS NOT NULL keeps NULL keys unmatched (as the equality JOIN did) even\n    -- under transform_null_in=1.\n    WHERE gpl.gene_id IN (\n        SELECT entrez_gene_id FROM gene\n        WHERE hugo_gene_symbol = 'ERBB2' AND entrez_gene_id IS NOT NULL)\n      AND stgp.alteration_type = multiIf(\n          'mutation' = 'mutation',           'MUTATION_EXTENDED',\n          'mutation' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'mutation' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'mutation' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n    UNION ALL\n    SELECT sample_unique_id, cancer_study_identifier\n    FROM sample_to_gene_panel_derived\n    WHERE gene_panel_id = 'WES'\n      AND alteration_type = multiIf(\n          'mutation' = 'mutation',           'MUTATION_EXTENDED',\n          'mutation' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'mutation' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'mutation' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n),\nprofiled AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT p.sample_unique_id) AS profiled_samples\n    FROM profiled_samples_for_gene p\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    GROUP BY sct.cancer_type\n)\nSELECT a.cancer_type,\n       a.altered_samples,\n       p.profiled_samples,\n       ROUND(a.altered_samples * 100.0 / NULLIF(p.profiled_samples, 0), 1) AS frequency_pct\nFROM altered a\nJOIN profiled p USING (cancer_type)\nWHERE p.profiled_samples >= 50"
  },
  {
    "name": "160-before-ERBB2-amplification",
    "source": "pr160",
    "sql": "SELECT * FROM gene_alteration_frequency_by_cancer_type(preference='all_studies_non_redundant', gene='ERBB2', alteration='amplification')"
  },
  {
    "name": "160-after-ERBB2-amplification",
    "source": "pr160",
    "sql": "WITH cohort AS (\n    SELECT cancer_study_identifier\n    FROM cancer_study_query_preferences\n    WHERE preference_name = 'all_studies_non_redundant'\n),\nsample_cancer_type AS (\n    SELECT cd.sample_unique_id, cd.attribute_value AS cancer_type\n    FROM clinical_data_derived cd\n    JOIN cohort c USING (cancer_study_identifier)\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\naltered AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT ged.sample_unique_id) AS altered_samples\n    FROM genomic_event_derived ged\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    WHERE ged.hugo_gene_symbol = 'ERBB2'\n      AND ged.off_panel = 0\n      AND (\n        ('amplification' = 'mutation'\n            AND ged.variant_type = 'mutation'\n            AND ged.mutation_status != 'UNCALLED')\n        OR ('amplification' = 'amplification'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = 2)\n        OR ('amplification' = 'deep_deletion'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = -2)\n        OR ('amplification' = 'structural_variant'\n            AND ged.variant_type = 'structural_variant')\n      )\n    GROUP BY sct.cancer_type\n),\nprofiled_samples_for_gene AS (\n    -- Map the user-facing alteration token to the alteration_type stored\n    -- on sample_to_gene_panel_derived. Same gene-in-panel-or-WES branch\n    -- as the mutation view, but with the matching alteration_type filter.\n    SELECT stgp.sample_unique_id, stgp.cancer_study_identifier\n    FROM sample_to_gene_panel_derived stgp\n    JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    -- IN-subquery instead of JOIN gene: resolves the symbol to its entrez\n    -- id(s) once, rather than joining every panel row against the gene\n    -- table before filtering. COUNT(DISTINCT) downstream makes the result\n    -- identical even for symbols with several (or duplicated) gene rows.\n    -- IS NOT NULL keeps NULL keys unmatched (as the equality JOIN did) even\n    -- under transform_null_in=1.\n    WHERE gpl.gene_id IN (\n        SELECT entrez_gene_id FROM gene\n        WHERE hugo_gene_symbol = 'ERBB2' AND entrez_gene_id IS NOT NULL)\n      AND stgp.alteration_type = multiIf(\n          'amplification' = 'mutation',           'MUTATION_EXTENDED',\n          'amplification' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'amplification' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'amplification' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n    UNION ALL\n    SELECT sample_unique_id, cancer_study_identifier\n    FROM sample_to_gene_panel_derived\n    WHERE gene_panel_id = 'WES'\n      AND alteration_type = multiIf(\n          'amplification' = 'mutation',           'MUTATION_EXTENDED',\n          'amplification' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'amplification' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'amplification' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n),\nprofiled AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT p.sample_unique_id) AS profiled_samples\n    FROM profiled_samples_for_gene p\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    GROUP BY sct.cancer_type\n)\nSELECT a.cancer_type,\n       a.altered_samples,\n       p.profiled_samples,\n       ROUND(a.altered_samples * 100.0 / NULLIF(p.profiled_samples, 0), 1) AS frequency_pct\nFROM altered a\nJOIN profiled p USING (cancer_type)\nWHERE p.profiled_samples >= 50"
  },
  {
    "name": "160-before-ERBB2-deep_deletion",
    "source": "pr160",
    "sql": "SELECT * FROM gene_alteration_frequency_by_cancer_type(preference='all_studies_non_redundant', gene='ERBB2', alteration='deep_deletion')"
  },
  {
    "name": "160-after-ERBB2-deep_deletion",
    "source": "pr160",
    "sql": "WITH cohort AS (\n    SELECT cancer_study_identifier\n    FROM cancer_study_query_preferences\n    WHERE preference_name = 'all_studies_non_redundant'\n),\nsample_cancer_type AS (\n    SELECT cd.sample_unique_id, cd.attribute_value AS cancer_type\n    FROM clinical_data_derived cd\n    JOIN cohort c USING (cancer_study_identifier)\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\naltered AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT ged.sample_unique_id) AS altered_samples\n    FROM genomic_event_derived ged\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    WHERE ged.hugo_gene_symbol = 'ERBB2'\n      AND ged.off_panel = 0\n      AND (\n        ('deep_deletion' = 'mutation'\n            AND ged.variant_type = 'mutation'\n            AND ged.mutation_status != 'UNCALLED')\n        OR ('deep_deletion' = 'amplification'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = 2)\n        OR ('deep_deletion' = 'deep_deletion'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = -2)\n        OR ('deep_deletion' = 'structural_variant'\n            AND ged.variant_type = 'structural_variant')\n      )\n    GROUP BY sct.cancer_type\n),\nprofiled_samples_for_gene AS (\n    -- Map the user-facing alteration token to the alteration_type stored\n    -- on sample_to_gene_panel_derived. Same gene-in-panel-or-WES branch\n    -- as the mutation view, but with the matching alteration_type filter.\n    SELECT stgp.sample_unique_id, stgp.cancer_study_identifier\n    FROM sample_to_gene_panel_derived stgp\n    JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    -- IN-subquery instead of JOIN gene: resolves the symbol to its entrez\n    -- id(s) once, rather than joining every panel row against the gene\n    -- table before filtering. COUNT(DISTINCT) downstream makes the result\n    -- identical even for symbols with several (or duplicated) gene rows.\n    -- IS NOT NULL keeps NULL keys unmatched (as the equality JOIN did) even\n    -- under transform_null_in=1.\n    WHERE gpl.gene_id IN (\n        SELECT entrez_gene_id FROM gene\n        WHERE hugo_gene_symbol = 'ERBB2' AND entrez_gene_id IS NOT NULL)\n      AND stgp.alteration_type = multiIf(\n          'deep_deletion' = 'mutation',           'MUTATION_EXTENDED',\n          'deep_deletion' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'deep_deletion' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'deep_deletion' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n    UNION ALL\n    SELECT sample_unique_id, cancer_study_identifier\n    FROM sample_to_gene_panel_derived\n    WHERE gene_panel_id = 'WES'\n      AND alteration_type = multiIf(\n          'deep_deletion' = 'mutation',           'MUTATION_EXTENDED',\n          'deep_deletion' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'deep_deletion' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'deep_deletion' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n),\nprofiled AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT p.sample_unique_id) AS profiled_samples\n    FROM profiled_samples_for_gene p\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    GROUP BY sct.cancer_type\n)\nSELECT a.cancer_type,\n       a.altered_samples,\n       p.profiled_samples,\n       ROUND(a.altered_samples * 100.0 / NULLIF(p.profiled_samples, 0), 1) AS frequency_pct\nFROM altered a\nJOIN profiled p USING (cancer_type)\nWHERE p.profiled_samples >= 50"
  },
  {
    "name": "160-before-ERBB2-structural_variant",
    "source": "pr160",
    "sql": "SELECT * FROM gene_alteration_frequency_by_cancer_type(preference='all_studies_non_redundant', gene='ERBB2', alteration='structural_variant')"
  },
  {
    "name": "160-after-ERBB2-structural_variant",
    "source": "pr160",
    "sql": "WITH cohort AS (\n    SELECT cancer_study_identifier\n    FROM cancer_study_query_preferences\n    WHERE preference_name = 'all_studies_non_redundant'\n),\nsample_cancer_type AS (\n    SELECT cd.sample_unique_id, cd.attribute_value AS cancer_type\n    FROM clinical_data_derived cd\n    JOIN cohort c USING (cancer_study_identifier)\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\naltered AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT ged.sample_unique_id) AS altered_samples\n    FROM genomic_event_derived ged\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    WHERE ged.hugo_gene_symbol = 'ERBB2'\n      AND ged.off_panel = 0\n      AND (\n        ('structural_variant' = 'mutation'\n            AND ged.variant_type = 'mutation'\n            AND ged.mutation_status != 'UNCALLED')\n        OR ('structural_variant' = 'amplification'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = 2)\n        OR ('structural_variant' = 'deep_deletion'\n            AND ged.variant_type = 'cna'\n            AND ged.cna_alteration = -2)\n        OR ('structural_variant' = 'structural_variant'\n            AND ged.variant_type = 'structural_variant')\n      )\n    GROUP BY sct.cancer_type\n),\nprofiled_samples_for_gene AS (\n    -- Map the user-facing alteration token to the alteration_type stored\n    -- on sample_to_gene_panel_derived. Same gene-in-panel-or-WES branch\n    -- as the mutation view, but with the matching alteration_type filter.\n    SELECT stgp.sample_unique_id, stgp.cancer_study_identifier\n    FROM sample_to_gene_panel_derived stgp\n    JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    -- IN-subquery instead of JOIN gene: resolves the symbol to its entrez\n    -- id(s) once, rather than joining every panel row against the gene\n    -- table before filtering. COUNT(DISTINCT) downstream makes the result\n    -- identical even for symbols with several (or duplicated) gene rows.\n    -- IS NOT NULL keeps NULL keys unmatched (as the equality JOIN did) even\n    -- under transform_null_in=1.\n    WHERE gpl.gene_id IN (\n        SELECT entrez_gene_id FROM gene\n        WHERE hugo_gene_symbol = 'ERBB2' AND entrez_gene_id IS NOT NULL)\n      AND stgp.alteration_type = multiIf(\n          'structural_variant' = 'mutation',           'MUTATION_EXTENDED',\n          'structural_variant' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'structural_variant' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'structural_variant' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n    UNION ALL\n    SELECT sample_unique_id, cancer_study_identifier\n    FROM sample_to_gene_panel_derived\n    WHERE gene_panel_id = 'WES'\n      AND alteration_type = multiIf(\n          'structural_variant' = 'mutation',           'MUTATION_EXTENDED',\n          'structural_variant' = 'amplification',      'COPY_NUMBER_ALTERATION',\n          'structural_variant' = 'deep_deletion',      'COPY_NUMBER_ALTERATION',\n          'structural_variant' = 'structural_variant', 'STRUCTURAL_VARIANT',\n          '')\n),\nprofiled AS (\n    SELECT sct.cancer_type,\n           COUNT(DISTINCT p.sample_unique_id) AS profiled_samples\n    FROM profiled_samples_for_gene p\n    JOIN cohort c USING (cancer_study_identifier)\n    JOIN sample_cancer_type sct USING (sample_unique_id)\n    GROUP BY sct.cancer_type\n)\nSELECT a.cancer_type,\n       a.altered_samples,\n       p.profiled_samples,\n       ROUND(a.altered_samples * 100.0 / NULLIF(p.profiled_samples, 0), 1) AS frequency_pct\nFROM altered a\nJOIN profiled p USING (cancer_type)\nWHERE p.profiled_samples >= 50"
  },
  {
    "name": "154-frequency-msk_chord_2024",
    "source": "pr154-fallback",
    "sql": "WITH\n        events AS (\n            SELECT sample_unique_id,\n                   (variant_type = 'mutation' AND mutation_status != 'UNCALLED') AS is_mutation,\n                   (variant_type = 'cna' AND cna_alteration = 2) AS is_amplification,\n                   (variant_type = 'cna' AND cna_alteration = -2) AS is_deep_deletion,\n                   (variant_type = 'structural_variant' AND mutation_status != 'UNCALLED') AS is_structural_variant\n            FROM genomic_event_derived\n            WHERE cancer_study_identifier IN ('msk_chord_2024')\n              AND hugo_gene_symbol IN ('TP53')\n              AND off_panel = 0\n        ),\n        profiled AS (\n            SELECT stgp.sample_unique_id AS sample_unique_id,\n                   stgp.alteration_type AS alteration_type\n            FROM sample_to_gene_panel_derived stgp\n            JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n            JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n            JOIN gene ge ON gpl.gene_id = ge.entrez_gene_id\n            WHERE stgp.cancer_study_identifier IN ('msk_chord_2024')\n              AND ge.hugo_gene_symbol IN ('TP53')\n              AND (stgp.alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT') OR (stgp.alteration_type = 'COPY_NUMBER_ALTERATION' AND stgp.genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE')))\n            UNION ALL\n            SELECT sample_unique_id, alteration_type\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('msk_chord_2024')\n              AND gene_panel_id = 'WES'\n              AND (alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT') OR (alteration_type = 'COPY_NUMBER_ALTERATION' AND genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE')))\n        )\n        SELECT *\n        FROM (\n            SELECT groupUniqArray(hugo_gene_symbol) AS matched_genes\n            FROM gene WHERE hugo_gene_symbol IN ('TP53')\n        ) AS gene_match\n        CROSS JOIN (\n            SELECT groupUniqArray(cancer_study_identifier) AS matched_studies\n            FROM cancer_study WHERE cancer_study_identifier IN ('msk_chord_2024')\n        ) AS study_match\n        CROSS JOIN (\n            SELECT uniqExactIf(sample_unique_id, is_mutation) AS altered_mutation,\n                   uniqExactIf(sample_unique_id, is_amplification) AS altered_amplification,\n                   uniqExactIf(sample_unique_id, is_deep_deletion) AS altered_deep_deletion,\n                   uniqExactIf(sample_unique_id, is_structural_variant)\n                       AS altered_structural_variant,\n                   uniqExactIf(sample_unique_id, is_mutation OR is_amplification\n                       OR is_deep_deletion OR is_structural_variant) AS altered_any\n            FROM events\n        ) AS altered\n        CROSS JOIN (\n            SELECT uniqExactIf(sample_unique_id, alteration_type = 'MUTATION_EXTENDED')\n                       AS profiled_MUTATION_EXTENDED,\n                   uniqExactIf(sample_unique_id, alteration_type = 'COPY_NUMBER_ALTERATION')\n                       AS profiled_COPY_NUMBER_ALTERATION,\n                   uniqExactIf(sample_unique_id, alteration_type = 'STRUCTURAL_VARIANT')\n                       AS profiled_STRUCTURAL_VARIANT,\n                   uniqExact(sample_unique_id) AS profiled_ANY\n            FROM profiled\n        ) AS profiled_counts"
  },
  {
    "name": "154-top-mutation-msk_chord_2024",
    "source": "pr154-fallback",
    "sql": "WITH\n        altered AS (\n            SELECT hugo_gene_symbol, COUNT(DISTINCT sample_unique_id) AS altered_samples\n            FROM genomic_event_derived\n            WHERE cancer_study_identifier IN ('msk_chord_2024')\n              AND off_panel = 0\n              AND (variant_type = 'mutation' AND mutation_status != 'UNCALLED')\n            GROUP BY hugo_gene_symbol\n            ORDER BY altered_samples DESC, hugo_gene_symbol ASC\n            LIMIT 20\n        ),\n        wes_samples AS (\n            SELECT DISTINCT sample_unique_id\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('msk_chord_2024')\n              AND gene_panel_id = 'WES'\n              AND (alteration_type IN ('MUTATION_EXTENDED'))\n        ),\n        panel_profiled AS (\n            SELECT ge.hugo_gene_symbol AS hugo_gene_symbol,\n                   COUNT(DISTINCT stgp.sample_unique_id) AS n\n            FROM sample_to_gene_panel_derived stgp\n            JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n            JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n            JOIN gene ge ON gpl.gene_id = ge.entrez_gene_id\n            WHERE stgp.cancer_study_identifier IN ('msk_chord_2024')\n              AND (stgp.alteration_type IN ('MUTATION_EXTENDED'))\n              AND stgp.sample_unique_id NOT IN (SELECT sample_unique_id FROM wes_samples)\n              AND ge.hugo_gene_symbol IN (SELECT hugo_gene_symbol FROM altered)\n            GROUP BY ge.hugo_gene_symbol\n        )\n        SELECT a.hugo_gene_symbol AS hugo_gene_symbol,\n               a.altered_samples AS altered_samples,\n               (SELECT count() FROM wes_samples) + COALESCE(p.n, 0) AS profiled_samples\n        FROM altered a\n        LEFT JOIN panel_profiled p ON a.hugo_gene_symbol = p.hugo_gene_symbol\n        ORDER BY altered_samples DESC, hugo_gene_symbol ASC"
  },
  {
    "name": "154-profiled-msk_chord_2024",
    "source": "pr154-fallback",
    "sql": "WITH\n        profiled AS (\n            SELECT cancer_study_identifier, alteration_type AS profile_type,\n                   sample_unique_id, gene_panel_id\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('msk_chord_2024')\n            UNION ALL\n            SELECT cancer_study_identifier, 'COPY_NUMBER_ALTERATION_DISCRETE' AS profile_type,\n                   sample_unique_id, gene_panel_id\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('msk_chord_2024')\n              AND ((alteration_type = 'COPY_NUMBER_ALTERATION' AND genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE')))\n            UNION ALL\n            SELECT cancer_study_identifier, 'ANY_MUT_CNA_SV' AS profile_type,\n                   sample_unique_id, gene_panel_id\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('msk_chord_2024')\n              AND (alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT') OR (alteration_type = 'COPY_NUMBER_ALTERATION' AND genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE')))\n        ),\n        sample_patient AS (\n            SELECT sample_unique_id, patient_unique_id\n            FROM sample_derived\n            WHERE cancer_study_identifier IN ('msk_chord_2024')\n        )\n        SELECT * FROM (\n            SELECT cancer_study_identifier, 'ALL_SAMPLES' AS profile_type,\n                   COUNT(DISTINCT sample_unique_id) AS samples,\n                   COUNT(DISTINCT patient_unique_id) AS patients,\n                   toUInt64(0) AS wes_samples\n            FROM sample_derived\n            WHERE cancer_study_identifier IN ('msk_chord_2024')\n            GROUP BY cancer_study_identifier\n            UNION ALL\n            SELECT p.cancer_study_identifier, p.profile_type,\n                   COUNT(DISTINCT p.sample_unique_id),\n                   COUNT(DISTINCT nullIf(sp.patient_unique_id, '')),\n                   COUNT(DISTINCT if(p.gene_panel_id = 'WES', p.sample_unique_id, NULL))\n            FROM profiled p\n            LEFT JOIN sample_patient sp ON p.sample_unique_id = sp.sample_unique_id\n            GROUP BY p.cancer_study_identifier, p.profile_type\n        )\n        ORDER BY cancer_study_identifier, profile_type = 'ALL_SAMPLES' DESC,\n                 profile_type = 'ANY_MUT_CNA_SV' DESC, samples DESC, profile_type"
  },
  {
    "name": "154-frequency-brca_tcga_pan_can_atlas_2018",
    "source": "pr154-fallback",
    "sql": "WITH\n        events AS (\n            SELECT sample_unique_id,\n                   (variant_type = 'mutation' AND mutation_status != 'UNCALLED') AS is_mutation,\n                   (variant_type = 'cna' AND cna_alteration = 2) AS is_amplification,\n                   (variant_type = 'cna' AND cna_alteration = -2) AS is_deep_deletion,\n                   (variant_type = 'structural_variant' AND mutation_status != 'UNCALLED') AS is_structural_variant\n            FROM genomic_event_derived\n            WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n              AND hugo_gene_symbol IN ('TP53')\n              AND off_panel = 0\n        ),\n        profiled AS (\n            SELECT stgp.sample_unique_id AS sample_unique_id,\n                   stgp.alteration_type AS alteration_type\n            FROM sample_to_gene_panel_derived stgp\n            JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n            JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n            JOIN gene ge ON gpl.gene_id = ge.entrez_gene_id\n            WHERE stgp.cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n              AND ge.hugo_gene_symbol IN ('TP53')\n              AND (stgp.alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT') OR (stgp.alteration_type = 'COPY_NUMBER_ALTERATION' AND stgp.genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE')))\n            UNION ALL\n            SELECT sample_unique_id, alteration_type\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n              AND gene_panel_id = 'WES'\n              AND (alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT') OR (alteration_type = 'COPY_NUMBER_ALTERATION' AND genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE')))\n        )\n        SELECT *\n        FROM (\n            SELECT groupUniqArray(hugo_gene_symbol) AS matched_genes\n            FROM gene WHERE hugo_gene_symbol IN ('TP53')\n        ) AS gene_match\n        CROSS JOIN (\n            SELECT groupUniqArray(cancer_study_identifier) AS matched_studies\n            FROM cancer_study WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n        ) AS study_match\n        CROSS JOIN (\n            SELECT uniqExactIf(sample_unique_id, is_mutation) AS altered_mutation,\n                   uniqExactIf(sample_unique_id, is_amplification) AS altered_amplification,\n                   uniqExactIf(sample_unique_id, is_deep_deletion) AS altered_deep_deletion,\n                   uniqExactIf(sample_unique_id, is_structural_variant)\n                       AS altered_structural_variant,\n                   uniqExactIf(sample_unique_id, is_mutation OR is_amplification\n                       OR is_deep_deletion OR is_structural_variant) AS altered_any\n            FROM events\n        ) AS altered\n        CROSS JOIN (\n            SELECT uniqExactIf(sample_unique_id, alteration_type = 'MUTATION_EXTENDED')\n                       AS profiled_MUTATION_EXTENDED,\n                   uniqExactIf(sample_unique_id, alteration_type = 'COPY_NUMBER_ALTERATION')\n                       AS profiled_COPY_NUMBER_ALTERATION,\n                   uniqExactIf(sample_unique_id, alteration_type = 'STRUCTURAL_VARIANT')\n                       AS profiled_STRUCTURAL_VARIANT,\n                   uniqExact(sample_unique_id) AS profiled_ANY\n            FROM profiled\n        ) AS profiled_counts"
  },
  {
    "name": "154-top-mutation-brca_tcga_pan_can_atlas_2018",
    "source": "pr154-fallback",
    "sql": "WITH\n        altered AS (\n            SELECT hugo_gene_symbol, COUNT(DISTINCT sample_unique_id) AS altered_samples\n            FROM genomic_event_derived\n            WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n              AND off_panel = 0\n              AND (variant_type = 'mutation' AND mutation_status != 'UNCALLED')\n            GROUP BY hugo_gene_symbol\n            ORDER BY altered_samples DESC, hugo_gene_symbol ASC\n            LIMIT 20\n        ),\n        wes_samples AS (\n            SELECT DISTINCT sample_unique_id\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n              AND gene_panel_id = 'WES'\n              AND (alteration_type IN ('MUTATION_EXTENDED'))\n        ),\n        panel_profiled AS (\n            SELECT ge.hugo_gene_symbol AS hugo_gene_symbol,\n                   COUNT(DISTINCT stgp.sample_unique_id) AS n\n            FROM sample_to_gene_panel_derived stgp\n            JOIN gene_panel gp ON stgp.gene_panel_id = gp.stable_id\n            JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n            JOIN gene ge ON gpl.gene_id = ge.entrez_gene_id\n            WHERE stgp.cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n              AND (stgp.alteration_type IN ('MUTATION_EXTENDED'))\n              AND stgp.sample_unique_id NOT IN (SELECT sample_unique_id FROM wes_samples)\n              AND ge.hugo_gene_symbol IN (SELECT hugo_gene_symbol FROM altered)\n            GROUP BY ge.hugo_gene_symbol\n        )\n        SELECT a.hugo_gene_symbol AS hugo_gene_symbol,\n               a.altered_samples AS altered_samples,\n               (SELECT count() FROM wes_samples) + COALESCE(p.n, 0) AS profiled_samples\n        FROM altered a\n        LEFT JOIN panel_profiled p ON a.hugo_gene_symbol = p.hugo_gene_symbol\n        ORDER BY altered_samples DESC, hugo_gene_symbol ASC"
  },
  {
    "name": "154-profiled-brca_tcga_pan_can_atlas_2018",
    "source": "pr154-fallback",
    "sql": "WITH\n        profiled AS (\n            SELECT cancer_study_identifier, alteration_type AS profile_type,\n                   sample_unique_id, gene_panel_id\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n            UNION ALL\n            SELECT cancer_study_identifier, 'COPY_NUMBER_ALTERATION_DISCRETE' AS profile_type,\n                   sample_unique_id, gene_panel_id\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n              AND ((alteration_type = 'COPY_NUMBER_ALTERATION' AND genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE')))\n            UNION ALL\n            SELECT cancer_study_identifier, 'ANY_MUT_CNA_SV' AS profile_type,\n                   sample_unique_id, gene_panel_id\n            FROM sample_to_gene_panel_derived\n            WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n              AND (alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT') OR (alteration_type = 'COPY_NUMBER_ALTERATION' AND genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE')))\n        ),\n        sample_patient AS (\n            SELECT sample_unique_id, patient_unique_id\n            FROM sample_derived\n            WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n        )\n        SELECT * FROM (\n            SELECT cancer_study_identifier, 'ALL_SAMPLES' AS profile_type,\n                   COUNT(DISTINCT sample_unique_id) AS samples,\n                   COUNT(DISTINCT patient_unique_id) AS patients,\n                   toUInt64(0) AS wes_samples\n            FROM sample_derived\n            WHERE cancer_study_identifier IN ('brca_tcga_pan_can_atlas_2018')\n            GROUP BY cancer_study_identifier\n            UNION ALL\n            SELECT p.cancer_study_identifier, p.profile_type,\n                   COUNT(DISTINCT p.sample_unique_id),\n                   COUNT(DISTINCT nullIf(sp.patient_unique_id, '')),\n                   COUNT(DISTINCT if(p.gene_panel_id = 'WES', p.sample_unique_id, NULL))\n            FROM profiled p\n            LEFT JOIN sample_patient sp ON p.sample_unique_id = sp.sample_unique_id\n            GROUP BY p.cancer_study_identifier, p.profile_type\n        )\n        ORDER BY cancer_study_identifier, profile_type = 'ALL_SAMPLES' DESC,\n                 profile_type = 'ANY_MUT_CNA_SV' DESC, samples DESC, profile_type"
  },
  {
    "name": "154-cancer-type-TP53",
    "source": "pr154-fallback",
    "sql": "SELECT * FROM (\n            SELECT cancer_type, 'TP53' AS hugo_gene_symbol,\n                   altered_samples, profiled_samples\n            FROM gene_alteration_frequency_by_cancer_type(\n                preference='all_studies_non_redundant', gene='TP53',\n                alteration='mutation')\n            )\n        ORDER BY altered_samples / profiled_samples DESC, altered_samples DESC, cancer_type ASC\n        LIMIT 20"
  },
  {
    "name": "154-cancer-type-TP53-pan-cancer-tcga",
    "source": "pr154-fallback",
    "sql": "SELECT * FROM (\n            SELECT cancer_type, 'TP53' AS hugo_gene_symbol,\n                   altered_samples, profiled_samples\n            FROM gene_alteration_frequency_by_cancer_type(\n                preference='pan_cancer_tcga', gene='TP53',\n                alteration='mutation')\n            )\n        ORDER BY altered_samples / profiled_samples DESC, altered_samples DESC, cancer_type ASC\n        LIMIT 20"
  },
  {
    "name": "154-build-study_gene_alteration_counts-msk_chord_2024",
    "source": "pr154-builds",
    "sql": "WITH\nalteration_to_profile AS (\n    SELECT t.1 AS alteration_type, t.2 AS profile_type\n    FROM (\n        SELECT arrayJoin([\n            ('mutation',           'MUTATION_EXTENDED'),\n            ('amplification',      'COPY_NUMBER_ALTERATION'),\n            ('deep_deletion',      'COPY_NUMBER_ALTERATION'),\n            ('structural_variant', 'STRUCTURAL_VARIANT'),\n            ('any',                'ANY')\n        ]) AS t\n    )\n),\nevents AS (\n    SELECT cancer_study_identifier,\n           hugo_gene_symbol,\n           sample_unique_id,\n           multiIf(\n               variant_type = 'mutation' AND mutation_status != 'UNCALLED', 'mutation',\n               variant_type = 'cna' AND cna_alteration = 2,                  'amplification',\n               variant_type = 'cna' AND cna_alteration = -2,                 'deep_deletion',\n               variant_type = 'structural_variant' AND mutation_status != 'UNCALLED',\n                                                                             'structural_variant',\n               '') AS event_type\n    FROM (SELECT * FROM genomic_event_derived WHERE cancer_study_identifier='msk_chord_2024')\n    WHERE off_panel = 0\n),\naltered AS (\n    -- event_type, not alteration_type: in ClickHouse a SELECT alias shadows\n    -- the source column in WHERE, so `'any' AS alteration_type` below would\n    -- turn a WHERE on alteration_type into a constant.\n    SELECT cancer_study_identifier, hugo_gene_symbol, event_type AS alteration_type,\n           COUNT(DISTINCT sample_unique_id) AS altered_samples,\n           COUNT(*) AS altered_events\n    FROM events\n    WHERE event_type != ''\n    GROUP BY cancer_study_identifier, hugo_gene_symbol, event_type\n    UNION ALL\n    SELECT cancer_study_identifier, hugo_gene_symbol, 'any' AS alteration_type,\n           COUNT(DISTINCT sample_unique_id) AS altered_samples,\n           COUNT(*) AS altered_events\n    FROM events\n    WHERE event_type != ''\n    GROUP BY cancer_study_identifier, hugo_gene_symbol\n),\ntyped_profile_rows AS (\n    -- Same profile filters as the {mutation,cna,sv}_*_coverage views:\n    -- CNA rows only from DISCRETE profiles.\n    SELECT cancer_study_identifier, alteration_type AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='msk_chord_2024')\n    WHERE alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT')\n       OR (alteration_type = 'COPY_NUMBER_ALTERATION'\n           AND genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE'))\n),\nprofile_rows AS (\n    SELECT cancer_study_identifier, profile_type, sample_unique_id, gene_panel_id\n    FROM typed_profile_rows\n    UNION ALL\n    SELECT cancer_study_identifier, 'ANY' AS profile_type, sample_unique_id, gene_panel_id\n    FROM typed_profile_rows\n),\nsample_buckets AS (\n    -- One row per sample per profile type: its panel-set signature.\n    SELECT cancer_study_identifier, profile_type, sample_unique_id,\n           arraySort(groupUniqArray(gene_panel_id)) AS panels\n    FROM profile_rows\n    GROUP BY cancer_study_identifier, profile_type, sample_unique_id\n),\nwes_profiled AS (\n    SELECT cancer_study_identifier, profile_type, COUNT(*) AS n\n    FROM sample_buckets\n    WHERE has(panels, 'WES')\n    GROUP BY cancer_study_identifier, profile_type\n),\nsignature_counts AS (\n    SELECT cancer_study_identifier, profile_type, panels, COUNT(*) AS n\n    FROM sample_buckets\n    WHERE NOT has(panels, 'WES')\n    GROUP BY cancer_study_identifier, profile_type, panels\n),\npanel_genes AS (\n    -- Same panel -> gene chain as mutation_panel_gene_coverage.\n    SELECT gp.stable_id AS gene_panel_id, g.hugo_gene_symbol AS hugo_gene_symbol\n    FROM gene_panel gp\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    JOIN gene g ON gpl.gene_id = g.entrez_gene_id\n),\npanel_profiled AS (\n    SELECT cancer_study_identifier, profile_type, hugo_gene_symbol, SUM(n) AS n\n    FROM (\n        -- DISTINCT: a gene listed by two panels of one signature counts once.\n        SELECT DISTINCT s.cancer_study_identifier, s.profile_type, s.panels, s.n, pg.hugo_gene_symbol\n        FROM (\n            SELECT cancer_study_identifier, profile_type, panels, n,\n                   arrayJoin(panels) AS gene_panel_id\n            FROM signature_counts\n        ) s\n        JOIN panel_genes pg ON s.gene_panel_id = pg.gene_panel_id\n    )\n    GROUP BY cancer_study_identifier, profile_type, hugo_gene_symbol\n)\nSELECT a.cancer_study_identifier,\n       a.hugo_gene_symbol,\n       a.alteration_type,\n       a.altered_samples,\n       COALESCE(w.n, 0) + COALESCE(p.n, 0) AS profiled_samples,\n       a.altered_events,\n       now() AS built_at\nFROM altered a\nJOIN alteration_to_profile m ON a.alteration_type = m.alteration_type\nLEFT JOIN wes_profiled w\n    ON w.cancer_study_identifier = a.cancer_study_identifier\n   AND w.profile_type = m.profile_type\nLEFT JOIN panel_profiled p\n    ON p.cancer_study_identifier = a.cancer_study_identifier\n   AND p.profile_type = m.profile_type\n   AND p.hugo_gene_symbol = a.hugo_gene_symbol"
  },
  {
    "name": "154-build-study_gene_alteration_counts-brca_tcga_pan_can_atlas_2018",
    "source": "pr154-builds",
    "sql": "WITH\nalteration_to_profile AS (\n    SELECT t.1 AS alteration_type, t.2 AS profile_type\n    FROM (\n        SELECT arrayJoin([\n            ('mutation',           'MUTATION_EXTENDED'),\n            ('amplification',      'COPY_NUMBER_ALTERATION'),\n            ('deep_deletion',      'COPY_NUMBER_ALTERATION'),\n            ('structural_variant', 'STRUCTURAL_VARIANT'),\n            ('any',                'ANY')\n        ]) AS t\n    )\n),\nevents AS (\n    SELECT cancer_study_identifier,\n           hugo_gene_symbol,\n           sample_unique_id,\n           multiIf(\n               variant_type = 'mutation' AND mutation_status != 'UNCALLED', 'mutation',\n               variant_type = 'cna' AND cna_alteration = 2,                  'amplification',\n               variant_type = 'cna' AND cna_alteration = -2,                 'deep_deletion',\n               variant_type = 'structural_variant' AND mutation_status != 'UNCALLED',\n                                                                             'structural_variant',\n               '') AS event_type\n    FROM (SELECT * FROM genomic_event_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\n    WHERE off_panel = 0\n),\naltered AS (\n    -- event_type, not alteration_type: in ClickHouse a SELECT alias shadows\n    -- the source column in WHERE, so `'any' AS alteration_type` below would\n    -- turn a WHERE on alteration_type into a constant.\n    SELECT cancer_study_identifier, hugo_gene_symbol, event_type AS alteration_type,\n           COUNT(DISTINCT sample_unique_id) AS altered_samples,\n           COUNT(*) AS altered_events\n    FROM events\n    WHERE event_type != ''\n    GROUP BY cancer_study_identifier, hugo_gene_symbol, event_type\n    UNION ALL\n    SELECT cancer_study_identifier, hugo_gene_symbol, 'any' AS alteration_type,\n           COUNT(DISTINCT sample_unique_id) AS altered_samples,\n           COUNT(*) AS altered_events\n    FROM events\n    WHERE event_type != ''\n    GROUP BY cancer_study_identifier, hugo_gene_symbol\n),\ntyped_profile_rows AS (\n    -- Same profile filters as the {mutation,cna,sv}_*_coverage views:\n    -- CNA rows only from DISCRETE profiles.\n    SELECT cancer_study_identifier, alteration_type AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\n    WHERE alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT')\n       OR (alteration_type = 'COPY_NUMBER_ALTERATION'\n           AND genetic_profile_id IN (SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE'))\n),\nprofile_rows AS (\n    SELECT cancer_study_identifier, profile_type, sample_unique_id, gene_panel_id\n    FROM typed_profile_rows\n    UNION ALL\n    SELECT cancer_study_identifier, 'ANY' AS profile_type, sample_unique_id, gene_panel_id\n    FROM typed_profile_rows\n),\nsample_buckets AS (\n    -- One row per sample per profile type: its panel-set signature.\n    SELECT cancer_study_identifier, profile_type, sample_unique_id,\n           arraySort(groupUniqArray(gene_panel_id)) AS panels\n    FROM profile_rows\n    GROUP BY cancer_study_identifier, profile_type, sample_unique_id\n),\nwes_profiled AS (\n    SELECT cancer_study_identifier, profile_type, COUNT(*) AS n\n    FROM sample_buckets\n    WHERE has(panels, 'WES')\n    GROUP BY cancer_study_identifier, profile_type\n),\nsignature_counts AS (\n    SELECT cancer_study_identifier, profile_type, panels, COUNT(*) AS n\n    FROM sample_buckets\n    WHERE NOT has(panels, 'WES')\n    GROUP BY cancer_study_identifier, profile_type, panels\n),\npanel_genes AS (\n    -- Same panel -> gene chain as mutation_panel_gene_coverage.\n    SELECT gp.stable_id AS gene_panel_id, g.hugo_gene_symbol AS hugo_gene_symbol\n    FROM gene_panel gp\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    JOIN gene g ON gpl.gene_id = g.entrez_gene_id\n),\npanel_profiled AS (\n    SELECT cancer_study_identifier, profile_type, hugo_gene_symbol, SUM(n) AS n\n    FROM (\n        -- DISTINCT: a gene listed by two panels of one signature counts once.\n        SELECT DISTINCT s.cancer_study_identifier, s.profile_type, s.panels, s.n, pg.hugo_gene_symbol\n        FROM (\n            SELECT cancer_study_identifier, profile_type, panels, n,\n                   arrayJoin(panels) AS gene_panel_id\n            FROM signature_counts\n        ) s\n        JOIN panel_genes pg ON s.gene_panel_id = pg.gene_panel_id\n    )\n    GROUP BY cancer_study_identifier, profile_type, hugo_gene_symbol\n)\nSELECT a.cancer_study_identifier,\n       a.hugo_gene_symbol,\n       a.alteration_type,\n       a.altered_samples,\n       COALESCE(w.n, 0) + COALESCE(p.n, 0) AS profiled_samples,\n       a.altered_events,\n       now() AS built_at\nFROM altered a\nJOIN alteration_to_profile m ON a.alteration_type = m.alteration_type\nLEFT JOIN wes_profiled w\n    ON w.cancer_study_identifier = a.cancer_study_identifier\n   AND w.profile_type = m.profile_type\nLEFT JOIN panel_profiled p\n    ON p.cancer_study_identifier = a.cancer_study_identifier\n   AND p.profile_type = m.profile_type\n   AND p.hugo_gene_symbol = a.hugo_gene_symbol"
  },
  {
    "name": "154-build-cancer_type_gene_alteration_counts-msk_chord_2024",
    "source": "pr154-builds",
    "sql": "WITH\nalteration_to_profile AS (\n    SELECT t.1 AS alteration_type, t.2 AS profile_type\n    FROM (\n        SELECT arrayJoin([\n            ('mutation',           'MUTATION_EXTENDED'),\n            ('amplification',      'COPY_NUMBER_ALTERATION'),\n            ('deep_deletion',      'COPY_NUMBER_ALTERATION'),\n            ('structural_variant', 'STRUCTURAL_VARIANT'),\n            ('any',                'ANY')\n        ]) AS t\n    )\n),\nsample_cancer_type AS (\n    SELECT q.preference_name AS preference_name,\n           cd.cancer_study_identifier AS cancer_study_identifier,\n           cd.sample_unique_id AS sample_unique_id,\n           cd.attribute_value AS cancer_type\n    FROM (SELECT * FROM clinical_data_derived WHERE cancer_study_identifier='msk_chord_2024') cd\n    JOIN cancer_study_query_preferences q ON cd.cancer_study_identifier = q.cancer_study_identifier\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\nevents AS (\n    SELECT cancer_study_identifier,\n           hugo_gene_symbol,\n           sample_unique_id,\n           multiIf(\n               variant_type = 'mutation' AND mutation_status != 'UNCALLED', 'mutation',\n               variant_type = 'cna' AND cna_alteration = 2,                  'amplification',\n               variant_type = 'cna' AND cna_alteration = -2,                 'deep_deletion',\n               variant_type = 'structural_variant' AND mutation_status != 'UNCALLED',\n                                                                             'structural_variant',\n               '') AS event_type\n    FROM (SELECT * FROM genomic_event_derived WHERE cancer_study_identifier='msk_chord_2024')\n    WHERE off_panel = 0\n      AND cancer_study_identifier IN (SELECT cancer_study_identifier FROM cancer_study_query_preferences)\n),\ntyped_events AS (\n    SELECT sct.preference_name AS preference_name, sct.cancer_type AS cancer_type,\n           e.hugo_gene_symbol AS hugo_gene_symbol, e.event_type AS event_type,\n           e.sample_unique_id AS sample_unique_id\n    FROM events e\n    JOIN sample_cancer_type sct\n        ON e.cancer_study_identifier = sct.cancer_study_identifier\n       AND e.sample_unique_id = sct.sample_unique_id\n    WHERE e.event_type != ''\n),\naltered AS (\n    SELECT preference_name, cancer_type, hugo_gene_symbol, event_type AS alteration_type,\n           COUNT(DISTINCT sample_unique_id) AS altered_samples\n    FROM typed_events\n    GROUP BY preference_name, cancer_type, hugo_gene_symbol, event_type\n    UNION ALL\n    SELECT preference_name, cancer_type, hugo_gene_symbol, 'any' AS alteration_type,\n           COUNT(DISTINCT sample_unique_id) AS altered_samples\n    FROM typed_events\n    GROUP BY preference_name, cancer_type, hugo_gene_symbol\n),\nprofile_rows AS (\n    SELECT cancer_study_identifier, alteration_type AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='msk_chord_2024')\n    WHERE alteration_type IN ('MUTATION_EXTENDED', 'COPY_NUMBER_ALTERATION', 'STRUCTURAL_VARIANT')\n    UNION ALL\n    SELECT cancer_study_identifier, 'ANY' AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='msk_chord_2024')\n    WHERE alteration_type IN ('MUTATION_EXTENDED', 'COPY_NUMBER_ALTERATION', 'STRUCTURAL_VARIANT')\n),\nsample_buckets AS (\n    SELECT sct.preference_name AS preference_name, sct.cancer_type AS cancer_type,\n           pr.profile_type AS profile_type, pr.sample_unique_id AS sample_unique_id,\n           arraySort(groupUniqArray(pr.gene_panel_id)) AS panels\n    FROM profile_rows pr\n    JOIN sample_cancer_type sct\n        ON pr.cancer_study_identifier = sct.cancer_study_identifier\n       AND pr.sample_unique_id = sct.sample_unique_id\n    GROUP BY preference_name, cancer_type, profile_type, sample_unique_id\n),\nwes_profiled AS (\n    SELECT preference_name, cancer_type, profile_type, COUNT(*) AS n\n    FROM sample_buckets\n    WHERE has(panels, 'WES')\n    GROUP BY preference_name, cancer_type, profile_type\n),\nsignature_counts AS (\n    SELECT preference_name, cancer_type, profile_type, panels, COUNT(*) AS n\n    FROM sample_buckets\n    WHERE NOT has(panels, 'WES')\n    GROUP BY preference_name, cancer_type, profile_type, panels\n),\npanel_genes AS (\n    SELECT gp.stable_id AS gene_panel_id, g.hugo_gene_symbol AS hugo_gene_symbol\n    FROM gene_panel gp\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    JOIN gene g ON gpl.gene_id = g.entrez_gene_id\n),\npanel_profiled AS (\n    SELECT preference_name, cancer_type, profile_type, hugo_gene_symbol, SUM(n) AS n\n    FROM (\n        SELECT DISTINCT s.preference_name, s.cancer_type, s.profile_type, s.panels, s.n,\n                        pg.hugo_gene_symbol\n        FROM (\n            SELECT preference_name, cancer_type, profile_type, panels, n,\n                   arrayJoin(panels) AS gene_panel_id\n            FROM signature_counts\n        ) s\n        JOIN panel_genes pg ON s.gene_panel_id = pg.gene_panel_id\n    )\n    GROUP BY preference_name, cancer_type, profile_type, hugo_gene_symbol\n)\nSELECT a.preference_name,\n       a.cancer_type,\n       a.hugo_gene_symbol,\n       a.alteration_type,\n       a.altered_samples,\n       COALESCE(w.n, 0) + COALESCE(p.n, 0) AS profiled_samples,\n       now() AS built_at\nFROM altered a\nJOIN alteration_to_profile m ON a.alteration_type = m.alteration_type\nLEFT JOIN wes_profiled w\n    ON w.preference_name = a.preference_name\n   AND w.cancer_type = a.cancer_type\n   AND w.profile_type = m.profile_type\nLEFT JOIN panel_profiled p\n    ON p.preference_name = a.preference_name\n   AND p.cancer_type = a.cancer_type\n   AND p.profile_type = m.profile_type\n   AND p.hugo_gene_symbol = a.hugo_gene_symbol"
  },
  {
    "name": "154-build-cancer_type_gene_alteration_counts-brca_tcga_pan_can_atlas_2018",
    "source": "pr154-builds",
    "sql": "WITH\nalteration_to_profile AS (\n    SELECT t.1 AS alteration_type, t.2 AS profile_type\n    FROM (\n        SELECT arrayJoin([\n            ('mutation',           'MUTATION_EXTENDED'),\n            ('amplification',      'COPY_NUMBER_ALTERATION'),\n            ('deep_deletion',      'COPY_NUMBER_ALTERATION'),\n            ('structural_variant', 'STRUCTURAL_VARIANT'),\n            ('any',                'ANY')\n        ]) AS t\n    )\n),\nsample_cancer_type AS (\n    SELECT q.preference_name AS preference_name,\n           cd.cancer_study_identifier AS cancer_study_identifier,\n           cd.sample_unique_id AS sample_unique_id,\n           cd.attribute_value AS cancer_type\n    FROM (SELECT * FROM clinical_data_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018') cd\n    JOIN cancer_study_query_preferences q ON cd.cancer_study_identifier = q.cancer_study_identifier\n    WHERE cd.attribute_name = 'CANCER_TYPE'\n),\nevents AS (\n    SELECT cancer_study_identifier,\n           hugo_gene_symbol,\n           sample_unique_id,\n           multiIf(\n               variant_type = 'mutation' AND mutation_status != 'UNCALLED', 'mutation',\n               variant_type = 'cna' AND cna_alteration = 2,                  'amplification',\n               variant_type = 'cna' AND cna_alteration = -2,                 'deep_deletion',\n               variant_type = 'structural_variant' AND mutation_status != 'UNCALLED',\n                                                                             'structural_variant',\n               '') AS event_type\n    FROM (SELECT * FROM genomic_event_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\n    WHERE off_panel = 0\n      AND cancer_study_identifier IN (SELECT cancer_study_identifier FROM cancer_study_query_preferences)\n),\ntyped_events AS (\n    SELECT sct.preference_name AS preference_name, sct.cancer_type AS cancer_type,\n           e.hugo_gene_symbol AS hugo_gene_symbol, e.event_type AS event_type,\n           e.sample_unique_id AS sample_unique_id\n    FROM events e\n    JOIN sample_cancer_type sct\n        ON e.cancer_study_identifier = sct.cancer_study_identifier\n       AND e.sample_unique_id = sct.sample_unique_id\n    WHERE e.event_type != ''\n),\naltered AS (\n    SELECT preference_name, cancer_type, hugo_gene_symbol, event_type AS alteration_type,\n           COUNT(DISTINCT sample_unique_id) AS altered_samples\n    FROM typed_events\n    GROUP BY preference_name, cancer_type, hugo_gene_symbol, event_type\n    UNION ALL\n    SELECT preference_name, cancer_type, hugo_gene_symbol, 'any' AS alteration_type,\n           COUNT(DISTINCT sample_unique_id) AS altered_samples\n    FROM typed_events\n    GROUP BY preference_name, cancer_type, hugo_gene_symbol\n),\nprofile_rows AS (\n    SELECT cancer_study_identifier, alteration_type AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\n    WHERE alteration_type IN ('MUTATION_EXTENDED', 'COPY_NUMBER_ALTERATION', 'STRUCTURAL_VARIANT')\n    UNION ALL\n    SELECT cancer_study_identifier, 'ANY' AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\n    WHERE alteration_type IN ('MUTATION_EXTENDED', 'COPY_NUMBER_ALTERATION', 'STRUCTURAL_VARIANT')\n),\nsample_buckets AS (\n    SELECT sct.preference_name AS preference_name, sct.cancer_type AS cancer_type,\n           pr.profile_type AS profile_type, pr.sample_unique_id AS sample_unique_id,\n           arraySort(groupUniqArray(pr.gene_panel_id)) AS panels\n    FROM profile_rows pr\n    JOIN sample_cancer_type sct\n        ON pr.cancer_study_identifier = sct.cancer_study_identifier\n       AND pr.sample_unique_id = sct.sample_unique_id\n    GROUP BY preference_name, cancer_type, profile_type, sample_unique_id\n),\nwes_profiled AS (\n    SELECT preference_name, cancer_type, profile_type, COUNT(*) AS n\n    FROM sample_buckets\n    WHERE has(panels, 'WES')\n    GROUP BY preference_name, cancer_type, profile_type\n),\nsignature_counts AS (\n    SELECT preference_name, cancer_type, profile_type, panels, COUNT(*) AS n\n    FROM sample_buckets\n    WHERE NOT has(panels, 'WES')\n    GROUP BY preference_name, cancer_type, profile_type, panels\n),\npanel_genes AS (\n    SELECT gp.stable_id AS gene_panel_id, g.hugo_gene_symbol AS hugo_gene_symbol\n    FROM gene_panel gp\n    JOIN gene_panel_list gpl ON gp.internal_id = gpl.internal_id\n    JOIN gene g ON gpl.gene_id = g.entrez_gene_id\n),\npanel_profiled AS (\n    SELECT preference_name, cancer_type, profile_type, hugo_gene_symbol, SUM(n) AS n\n    FROM (\n        SELECT DISTINCT s.preference_name, s.cancer_type, s.profile_type, s.panels, s.n,\n                        pg.hugo_gene_symbol\n        FROM (\n            SELECT preference_name, cancer_type, profile_type, panels, n,\n                   arrayJoin(panels) AS gene_panel_id\n            FROM signature_counts\n        ) s\n        JOIN panel_genes pg ON s.gene_panel_id = pg.gene_panel_id\n    )\n    GROUP BY preference_name, cancer_type, profile_type, hugo_gene_symbol\n)\nSELECT a.preference_name,\n       a.cancer_type,\n       a.hugo_gene_symbol,\n       a.alteration_type,\n       a.altered_samples,\n       COALESCE(w.n, 0) + COALESCE(p.n, 0) AS profiled_samples,\n       now() AS built_at\nFROM altered a\nJOIN alteration_to_profile m ON a.alteration_type = m.alteration_type\nLEFT JOIN wes_profiled w\n    ON w.preference_name = a.preference_name\n   AND w.cancer_type = a.cancer_type\n   AND w.profile_type = m.profile_type\nLEFT JOIN panel_profiled p\n    ON p.preference_name = a.preference_name\n   AND p.cancer_type = a.cancer_type\n   AND p.profile_type = m.profile_type\n   AND p.hugo_gene_symbol = a.hugo_gene_symbol"
  },
  {
    "name": "154-build-study_profiled_counts-msk_chord_2024",
    "source": "pr154-builds",
    "sql": "WITH\nsample_patient AS (\n    SELECT sample_unique_id, patient_unique_id\n    FROM (SELECT * FROM sample_derived WHERE cancer_study_identifier='msk_chord_2024')\n),\ndiscrete_cna_profiles AS (\n    SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE'\n),\nprofiled AS (\n    SELECT cancer_study_identifier, alteration_type AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='msk_chord_2024')\n    UNION ALL\n    SELECT cancer_study_identifier, 'COPY_NUMBER_ALTERATION_DISCRETE' AS profile_type,\n           sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='msk_chord_2024')\n    WHERE alteration_type = 'COPY_NUMBER_ALTERATION'\n      AND genetic_profile_id IN (SELECT stable_id FROM discrete_cna_profiles)\n    UNION ALL\n    SELECT cancer_study_identifier, 'ANY_MUT_CNA_SV' AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='msk_chord_2024')\n    WHERE alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT')\n       OR (alteration_type = 'COPY_NUMBER_ALTERATION'\n           AND genetic_profile_id IN (SELECT stable_id FROM discrete_cna_profiles))\n)\nSELECT cancer_study_identifier,\n       'ALL_SAMPLES' AS profile_type,\n       COUNT(DISTINCT sample_unique_id) AS samples,\n       COUNT(DISTINCT patient_unique_id) AS patients,\n       toUInt64(0) AS wes_samples,\n       now() AS built_at\nFROM (SELECT * FROM sample_derived WHERE cancer_study_identifier='msk_chord_2024')\nGROUP BY cancer_study_identifier\nUNION ALL\nSELECT p.cancer_study_identifier,\n       p.profile_type,\n       COUNT(DISTINCT p.sample_unique_id) AS samples,\n       COUNT(DISTINCT nullIf(sp.patient_unique_id, '')) AS patients,\n       COUNT(DISTINCT if(p.gene_panel_id = 'WES', p.sample_unique_id, NULL)) AS wes_samples,\n       now() AS built_at\nFROM profiled p\nLEFT JOIN sample_patient sp ON p.sample_unique_id = sp.sample_unique_id\nGROUP BY p.cancer_study_identifier, p.profile_type"
  },
  {
    "name": "154-build-study_profiled_counts-brca_tcga_pan_can_atlas_2018",
    "source": "pr154-builds",
    "sql": "WITH\nsample_patient AS (\n    SELECT sample_unique_id, patient_unique_id\n    FROM (SELECT * FROM sample_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\n),\ndiscrete_cna_profiles AS (\n    SELECT stable_id FROM genetic_profile WHERE datatype = 'DISCRETE'\n),\nprofiled AS (\n    SELECT cancer_study_identifier, alteration_type AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\n    UNION ALL\n    SELECT cancer_study_identifier, 'COPY_NUMBER_ALTERATION_DISCRETE' AS profile_type,\n           sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\n    WHERE alteration_type = 'COPY_NUMBER_ALTERATION'\n      AND genetic_profile_id IN (SELECT stable_id FROM discrete_cna_profiles)\n    UNION ALL\n    SELECT cancer_study_identifier, 'ANY_MUT_CNA_SV' AS profile_type, sample_unique_id, gene_panel_id\n    FROM (SELECT * FROM sample_to_gene_panel_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\n    WHERE alteration_type IN ('MUTATION_EXTENDED', 'STRUCTURAL_VARIANT')\n       OR (alteration_type = 'COPY_NUMBER_ALTERATION'\n           AND genetic_profile_id IN (SELECT stable_id FROM discrete_cna_profiles))\n)\nSELECT cancer_study_identifier,\n       'ALL_SAMPLES' AS profile_type,\n       COUNT(DISTINCT sample_unique_id) AS samples,\n       COUNT(DISTINCT patient_unique_id) AS patients,\n       toUInt64(0) AS wes_samples,\n       now() AS built_at\nFROM (SELECT * FROM sample_derived WHERE cancer_study_identifier='brca_tcga_pan_can_atlas_2018')\nGROUP BY cancer_study_identifier\nUNION ALL\nSELECT p.cancer_study_identifier,\n       p.profile_type,\n       COUNT(DISTINCT p.sample_unique_id) AS samples,\n       COUNT(DISTINCT nullIf(sp.patient_unique_id, '')) AS patients,\n       COUNT(DISTINCT if(p.gene_panel_id = 'WES', p.sample_unique_id, NULL)) AS wes_samples,\n       now() AS built_at\nFROM profiled p\nLEFT JOIN sample_patient sp ON p.sample_unique_id = sp.sample_unique_id\nGROUP BY p.cancer_study_identifier, p.profile_type"
  }
]
