diff --git a/clickhouse_search/backend/functions.py b/clickhouse_search/backend/functions.py
index 85e242665e..07329222b7 100644
--- a/clickhouse_search/backend/functions.py
+++ b/clickhouse_search/backend/functions.py
@@ -95,6 +95,13 @@ def process_rhs(self, compiler, connection):
return rhs_params[0], []
+@NestedField.register_lookup
+@ArrayField.register_lookup
+class ArrayAll(ArrayExists):
+ lookup_name = "array_all"
+ function = "arrayAll"
+
+
class ArrayFilter(lookups.Transform):
def __init__(self, *args, conditions=None, **kwargs):
super().__init__(*args, **kwargs)
@@ -113,6 +120,20 @@ class ArrayNotEmptyTransform(lookups.Transform):
output_field = BooleanField()
+@ArrayField.register_lookup
+class BitmapHasAny(ArrayLookup):
+ lookup_name = "bitmap_has_any"
+ function = "bitmapHasAny"
+
+ def process_lhs(self, compiler, connection):
+ lhs, lhs_params = super().process_lhs(compiler, connection)
+ return f'bitmapBuild({lhs})', lhs_params
+
+ def process_rhs(self, compiler, connection):
+ rhs, rhs_params = super().process_rhs(compiler, connection)
+ return f'bitmapBuild{rhs.split("::")[0]}', rhs_params
+
+
class DictGet(Func):
function = 'dictGet'
template = '%(function)s("%(dict_name)s", (%(fields)s), %(expressions)s)'
diff --git a/clickhouse_search/fixtures/clickhouse_saved_variants.json b/clickhouse_search/fixtures/clickhouse_saved_variants.json
index 8ad858a65f..6a7e7706e7 100644
--- a/clickhouse_search/fixtures/clickhouse_saved_variants.json
+++ b/clickhouse_search/fixtures/clickhouse_saved_variants.json
@@ -66,6 +66,8 @@
"sample_type": "WGS",
"xpos": 1248367227,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [6, 48],
"filters": [],
"sign": 1,
"calls": [
@@ -118,6 +120,8 @@
"sample_type": "WES",
"xpos": 2103343353,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [48, 45],
"filters": [],
"sign": 1,
"calls": [
@@ -196,6 +200,8 @@
"sample_type": "WGS",
"xpos": 1248367227,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": false,
+ "geneId_ids": [],
"filters": [],
"sign": 1,
"calls": [
diff --git a/clickhouse_search/fixtures/clickhouse_search.json b/clickhouse_search/fixtures/clickhouse_search.json
index c2f05942bc..55afe33016 100644
--- a/clickhouse_search/fixtures/clickhouse_search.json
+++ b/clickhouse_search/fixtures/clickhouse_search.json
@@ -274,6 +274,8 @@
"sample_type": "WES",
"xpos": 1000010439,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": false,
+ "geneId_ids": [],
"filters": [],
"sign": 1,
"calls": [
@@ -292,6 +294,8 @@
"sample_type": "WES",
"xpos": 1038724419,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [60, 72],
"filters": [],
"sign": 1,
"calls": [
@@ -310,6 +314,8 @@
"sample_type": "WES",
"xpos": 1091502721,
"is_gnomad_gt_5_percent": true,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [61, 60],
"filters": [],
"sign": 1,
"calls": [
@@ -328,6 +334,70 @@
"sample_type": "WES",
"xpos": 1091511686,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [61],
+ "filters": [
+ "VQSRTrancheSNP99.95to100.00"
+ ],
+ "sign": 1,
+ "calls": [
+ ["HG00733", 0, 0, 0, 45],
+ ["HG00732", 0, 0, 0, 24],
+ ["HG00731", 1, 58, 0.17241, 29]
+ ]
+ }
+}, {
+ "model": "clickhouse_search.entriessnvindel",
+ "pk": 52,
+ "fields": {
+ "key": 2,
+ "project_guid": "R0001_1kg",
+ "family_guid": "F000002_2_x",
+ "sample_type": "WES",
+ "xpos": 1038724419,
+ "is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [60, 72],
+ "filters": [],
+ "sign": 1,
+ "calls": [
+ ["HG00733", 0, 40, 0, 33],
+ ["HG00732", 1, 99, 0.625, 32],
+ ["HG00731", 2, 99, 1, 36]
+ ]
+ }
+}, {
+ "model": "clickhouse_search.entriessnvindel",
+ "pk": 53,
+ "fields": {
+ "key": 3,
+ "project_guid": "R0001_1kg",
+ "family_guid": "F000002_2_x",
+ "sample_type": "WES",
+ "xpos": 1091502721,
+ "is_gnomad_gt_5_percent": true,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [61, 60],
+ "filters": [],
+ "sign": 1,
+ "calls": [
+ ["HG00733", 1, 99, 0.40741, 27],
+ ["HG00732", 0, 99, 0.45946, 37],
+ ["HG00731", 1, 99, 1, 40]
+ ]
+ }
+}, {
+ "model": "clickhouse_search.entriessnvindel",
+ "pk": 54,
+ "fields": {
+ "key": 4,
+ "project_guid": "R0001_1kg",
+ "family_guid": "F000002_2_x",
+ "sample_type": "WES",
+ "xpos": 1091511686,
+ "is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [61],
"filters": [
"VQSRTrancheSNP99.95to100.00"
],
@@ -348,10 +418,11 @@
"sample_type": "WES",
"xpos": 1091502721,
"is_gnomad_gt_5_percent": true,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [61, 60],
"filters": [],
"sign": 1,
"calls": [
- ["NA20872", 1, 99, 0.34285714285714286, 35],
["NA20870", 1, 99, 0.6785714285714286, 28]
]
}
@@ -365,6 +436,8 @@
"sample_type": "WES",
"xpos": 1000010439,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": false,
+ "geneId_ids": [],
"filters": [],
"sign": 1,
"calls": [
@@ -381,6 +454,8 @@
"sample_type": "WES",
"xpos": 1038724419,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [60, 72],
"filters": [],
"sign": 1,
"calls": [
@@ -397,6 +472,8 @@
"sample_type": "WES",
"xpos": 1000010146,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": false,
+ "geneId_ids": [],
"filters": [],
"sign": 1,
"calls": [
@@ -413,6 +490,8 @@
"sample_type": "WGS",
"xpos": 1000010439,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": false,
+ "geneId_ids": [],
"filters": [],
"sign": 1,
"calls": [
@@ -429,6 +508,8 @@
"sample_type": "WGS",
"xpos": 1038724419,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [60, 72],
"filters": [],
"sign": 1,
"calls": [
@@ -445,6 +526,8 @@
"sample_type": "WGS",
"xpos": 1000010146,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": false,
+ "geneId_ids": [],
"filters": [],
"sign": 1,
"calls": [
@@ -461,6 +544,8 @@
"sample_type": "WGS",
"xpos": 1000010439,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": false,
+ "geneId_ids": [],
"filters": [],
"sign": 1,
"calls": [
@@ -479,6 +564,8 @@
"sample_type": "WGS",
"xpos": 1038724419,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [60, 72],
"filters": [],
"sign": 1,
"calls": [
@@ -497,6 +584,8 @@
"sample_type": "WGS",
"xpos": 1091502721,
"is_gnomad_gt_5_percent": true,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [61, 60],
"filters": [],
"sign": 1,
"calls": [
@@ -515,6 +604,8 @@
"sample_type": "WGS",
"xpos": 1091511686,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [61],
"filters": [
"VQSRTrancheSNP99.95to100.00"
],
@@ -535,6 +626,8 @@
"sample_type": "WGS",
"xpos": 1009310123,
"is_gnomad_gt_5_percent": false,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [71],
"filters": [],
"sign": 1,
"calls": [
@@ -581,6 +674,8 @@
"sample_type": "WES",
"xpos": 7143270172,
"is_gnomad_gt_5_percent": true,
+ "is_annotated_in_any_gene": true,
+ "geneId_ids": [73, 74],
"filters": [
"VQSRTrancheSNP99.90to99.95"
],
@@ -600,6 +695,8 @@
"sample_type": "WGS",
"xpos": 1000010439,
"is_gnomad_gt_5_percent": true,
+ "is_annotated_in_any_gene": false,
+ "geneId_ids": [],
"filters": [],
"sign": 1,
"calls": [
@@ -1116,6 +1213,66 @@
["HG00733", 1, 3, 29, false, 38721781, 38734440, 7, ["ENSG00000275023", "ENSG00000277258", "ENSG00000277972"], false, true, false]
]
}
+}, {
+ "model": "clickhouse_search.entriesgcnv",
+ "pk": 48,
+ "fields": {
+ "key": 16,
+ "project_guid": "R0001_1kg",
+ "family_guid": "F000002_2_x",
+ "xpos": 14022417556,
+ "filters": [],
+ "sign": 1,
+ "calls": [
+ ["HG00731", 1, 3, 38, false, 22438910, 22469796, 0, [], false, true, false],
+ ["HG00733", null, null, null, null, null, null, null, [], null, null, null]
+ ]
+ }
+}, {
+ "model": "clickhouse_search.entriesgcnv",
+ "pk": 49,
+ "fields": {
+ "key": 17,
+ "project_guid": "R0001_1kg",
+ "family_guid": "F000002_2_x",
+ "xpos": 16029802672,
+ "filters": [],
+ "sign": 1,
+ "calls": [
+ ["HG00731", 1, 3, 29, false, 29809156, 29815990, 8, ["ENSG00000103495", "ENSG00000167371", "ENSG00000280893"], false, true, false],
+ ["HG00733", 1, 3, 37, false, 29809156, 29815990, 8, ["ENSG00000103495", "ENSG00000167371", "ENSG00000280893"], false, true, false]
+ ]
+ }
+}, {
+ "model": "clickhouse_search.entriesgcnv",
+ "pk": 40,
+ "fields": {
+ "key": 18,
+ "project_guid": "R0001_1kg",
+ "family_guid": "F000002_2_x",
+ "xpos": 17038717327,
+ "filters": [],
+ "sign": 1,
+ "calls": [
+ ["HG00731", 2, 4, 13, true, 38717327, 38719636, 3, ["ENSG00000275023"], true, false, false],
+ ["HG00733", null, null, null, null, null, null, null, [], null, null, null]
+ ]
+ }
+}, {
+ "model": "clickhouse_search.entriesgcnv",
+ "pk": 41,
+ "fields": {
+ "key": 19,
+ "project_guid": "R0001_1kg",
+ "family_guid": "F000002_2_x",
+ "xpos": 17038721781,
+ "filters": [],
+ "sign": 1,
+ "calls": [
+ ["HG00731", 1, 3, 28, false, 38721781, 38735703, 7, ["ENSG00000275023", "ENSG00000277258", "ENSG00000277972"], false, true, false],
+ ["HG00733", 1, 3, 29, false, 38721781, 38734440, 7, ["ENSG00000275023", "ENSG00000277258", "ENSG00000277972"], false, true, false]
+ ]
+ }
}, {
"model": "clickhouse_search.entriesgcnv",
"pk": 32,
@@ -1339,5 +1496,35 @@
"gencode_gene_type": "protein_coding",
"gencode_release": 27
}
+},
+{
+ "model": "reference_data.geneinfo",
+ "pk": 73,
+ "fields": {
+ "gene_id": "ENSG00000271079",
+ "gene_symbol": "CTAGE15",
+ "chrom_grch38": "7",
+ "start_grch38": 143571801,
+ "end_grch38": 143574387,
+ "strand_grch38": "+",
+ "coding_region_size_grch38": 0,
+ "gencode_gene_type": "protein_coding",
+ "gencode_release": 27
+ }
+},
+{
+ "model": "reference_data.geneinfo",
+ "pk": 74,
+ "fields": {
+ "gene_id": "ENSG00000176227",
+ "gene_symbol": "CTAGE6",
+ "chrom_grch38": "7",
+ "start_grch38": 91500851,
+ "end_grch38": 91525764,
+ "strand_grch38": "+",
+ "coding_region_size_grch38": 0,
+ "gencode_gene_type": "protein_coding",
+ "gencode_release": 27
+ }
}
]
\ No newline at end of file
diff --git a/clickhouse_search/management/commands/set_saved_variant_key.py b/clickhouse_search/management/commands/set_saved_variant_key.py
index bc3ea807d2..07df47e39b 100644
--- a/clickhouse_search/management/commands/set_saved_variant_key.py
+++ b/clickhouse_search/management/commands/set_saved_variant_key.py
@@ -14,6 +14,7 @@
logger = logging.getLogger(__name__)
+BATCH_SIZE = 10000
GCNV_CALLSET_PATH = 'gs://seqr-datasets-gcnv/GRCh38/RDG_WES_Broad_Internal/v4/CMG_gCNV_2022_annotated.ensembl.round2_3.strvctvre.tsv.gz'
SV_ID_UPDATE_MAP = {
@@ -2257,23 +2258,30 @@ def _set_variant_keys(variants_ids, dataset_type, genome_version=GENOME_VERSION_
if not variant_key_map:
return set(variants_ids)
- mapped_variant_ids = variant_key_map.keys()
+ mapped_variant_ids = list(variant_key_map.keys())
if variant_id_updates:
reverse_lookup = {v: k for k, v in variant_id_updates.items()}
mapped_variant_ids = [reverse_lookup[vid] for vid in mapped_variant_ids]
- saved_variants = SavedVariant.objects.filter(
- family__project__genome_version=genome_version, variant_id__in=mapped_variant_ids,
- )
- for variant in saved_variants:
- if variant_id_updates:
- variant.variant_id = variant_id_updates[variant.variant_id]
- variant.key = variant_key_map[variant.variant_id]
- variant.dataset_type = dataset_type
+
update_fields = ['key', 'dataset_type']
if variant_id_updates:
update_fields.append('variant_id')
- num_updated = SavedVariant.objects.bulk_update(saved_variants, update_fields, batch_size=10000)
- logger.info(f'Updated keys for {num_updated} {dataset_type} (GRCh{genome_version}) variants')
+ total_num_updated = 0
+ for i in range(0, len(mapped_variant_ids), BATCH_SIZE):
+ batch_ids = mapped_variant_ids[i:i + BATCH_SIZE]
+ saved_variants = SavedVariant.objects.filter(
+ family__project__genome_version=genome_version, variant_id__in=batch_ids,
+ )
+ for variant in saved_variants:
+ if variant_id_updates:
+ variant.variant_id = variant_id_updates[variant.variant_id]
+ variant.key = variant_key_map[variant.variant_id]
+ variant.dataset_type = dataset_type
+ num_updated = SavedVariant.objects.bulk_update(saved_variants, update_fields)
+ logger.info(f'Updated batch of {num_updated}')
+ total_num_updated += num_updated
+
+ logger.info(f'Updated keys for {total_num_updated} {dataset_type} (GRCh{genome_version}) variants')
no_key = set(variants_ids) - set(variant_key_map.keys())
if no_key:
diff --git a/clickhouse_search/management/tests/set_saved_variant_key_tests.py b/clickhouse_search/management/tests/set_saved_variant_key_tests.py
index 56029887ee..be47bf5bb5 100644
--- a/clickhouse_search/management/tests/set_saved_variant_key_tests.py
+++ b/clickhouse_search/management/tests/set_saved_variant_key_tests.py
@@ -22,6 +22,7 @@ def setUpTestData(cls):
Sample.objects.filter(guid='S000154_na20889').update(dataset_type='SV', is_active=True)
SavedVariant.objects.update(key=None)
+ @mock.patch('clickhouse_search.management.commands.set_saved_variant_key.BATCH_SIZE', 2)
@mock.patch('seqr.utils.file_utils.subprocess.Popen')
def test_command(self, mock_subprocess):
mock_subprocess.return_value.stdout = self.MOCK_GCNV_DATA
@@ -31,9 +32,11 @@ def test_command(self, mock_subprocess):
('Updated genotypes for 7 variants', None),
('Finding keys for 1 MITO (GRCh38) variant ids', None),
('Found 1 keys', None),
+ ('Updated batch of 1', None),
('Updated keys for 1 MITO (GRCh38) variants', None),
('Finding keys for 1 SNV_INDEL (GRCh38) variant ids', None),
('Found 1 keys', None),
+ ('Updated batch of 2', None),
('Updated keys for 2 SNV_INDEL (GRCh38) variants', None),
('Finding keys for 2 SV_WGS (GRCh38) variant ids', None),
('Found 0 keys', None),
@@ -44,13 +47,16 @@ def test_command(self, mock_subprocess):
('Mapping reloaded SV_WES IDs to latest version', None),
('Finding keys for 1 SV_WES (GRCh38) variant ids', None),
('Found 1 keys', None),
+ ('Updated batch of 1', None),
('Updated keys for 1 SV_WES (GRCh38) variants', None),
('Mapping reloaded SV_WGS IDs to latest version', None),
('Finding keys for 1 SV_WGS (GRCh38) variant ids', None),
('Found 1 keys', None),
+ ('Updated batch of 1', None),
('Updated keys for 1 SV_WGS (GRCh38) variants', None),
('Finding keys for 7 SNV_INDEL (GRCh37) variant ids', None),
('Found 1 keys', None),
+ ('Updated batch of 1', None),
('Updated keys for 1 SNV_INDEL (GRCh37) variants', None),
('No key found for 6 variants', None),
('6 variants have no key, 0 of which have no search data, 6 of which are absent from the hail backend.', None),
@@ -109,7 +115,7 @@ def test_command(self, mock_subprocess):
('Finding keys for 2 SNV_INDEL (GRCh38) variant ids', None),
('Found 0 keys', None),
('3 variants have no key, 1 of which have no search data, 1 of which are absent from the hail backend.', None),
- ('1 remaining variants: M-14783-T-C - 14', None),
+ ('1 remaining variants: M-14783-T-C - fam14', None),
('Finding keys for 2 SV_WGS (GRCh38) variant ids', None),
('Found 0 keys', None),
('Finding keys for 2 SV_WES (GRCh38) variant ids', None),
diff --git a/clickhouse_search/managers.py b/clickhouse_search/managers.py
index 3012a33e2a..199d83b2d5 100644
--- a/clickhouse_search/managers.py
+++ b/clickhouse_search/managers.py
@@ -134,7 +134,7 @@ def transcript_fields(self):
@property
def single_sample_type(self):
- return getattr(self.model, 'SV_TYPES', None)
+ return getattr(self.entry_model, 'SAMPLE_TYPE', None)
@property
def entry_field(self):
@@ -156,11 +156,7 @@ def clinvar_field_prefix(self):
def subquery_join(self, subquery, join_key='key'):
- # Add key to intermediate select if not already present
join_field = next(field for field in subquery.model._meta.fields if field.name == join_key)
- if join_key not in subquery.query.values_select:
- subquery.query.values_select = tuple([join_key, *subquery.query.values_select])
- subquery.query.select = tuple([Col(subquery.model._meta.db_table, join_field), *subquery.query.select])
# Add the join operation to the query
table = SubqueryTable(subquery)
@@ -203,7 +199,7 @@ def _get_join_query_values(self, query, alias, conditional_selects):
query_select = query.annotation_values
for select_func in (conditional_selects or []):
query_select.update(select_func(query, prefix=f'{alias}_'))
- annotation_fields = query.annotation_fields + self.ENTRY_FIELDS
+ annotation_fields = query.annotation_fields
return query.values(
**{f'{alias}_{field}': F(field) for field in annotation_fields if field not in query_select},
**{f'{alias}_{field}': value for field, value in query_select.items()},
@@ -312,14 +308,14 @@ def search_compound_hets(self, primary_q, secondary_q):
secondary_q = secondary_q.explode_gene_id(secondary_gene_field)
conditional_fields = lambda query, **kwargs: {
- field: F(field) for field in [self.SELECTED_GENE_FIELD, 'clinvar', 'family_carriers', 'carriers', 'has_hom_alt', 'no_hom_alt_families']
+ field: F(field) for field in [self.SELECTED_GENE_FIELD, 'clinvar', 'family_carriers', 'carriers', 'has_hom_alt', 'no_hom_alt_families', 'familyGenotypes'] + self.ENTRY_FIELDS
if field in query.query.annotations
}
results = self.cross_join(
query=primary_q, alias='primary', join_query=secondary_q, join_alias='secondary',
conditional_selects=[
- self._conditional_selected_transcript_values, self._genotype_override_values, conditional_fields,
+ conditional_fields, self._conditional_selected_transcript_values, self._genotype_override_values,
],
)
return results.filter(
@@ -366,7 +362,13 @@ def _filter_frequency(self, results, freqs=None, pathogenicity=None, **kwargs):
}))
results = results.filter(af_q)
elif pop_filter.get('ac') is not None:
- results = results.filter(**{f'populations__{population}__ac__lte': pop_filter['ac']})
+ if 'ac' in pop_subfields:
+ ac_field = f'populations__{population}__ac'
+ else:
+ ac_field = f'ac_{population}'
+ results = results.annotate(**{ac_field: F(f'populations__{population}__het') + (F(f'populations__{population}__hom') * 2)})
+
+ results = results.filter(**{f'{ac_field}__lte': pop_filter['ac']})
if pop_filter.get('hh') is not None:
for subfield in ['hom', 'hemi']:
@@ -474,9 +476,9 @@ def filter_annotations(self, results, annotations=None, pathogenicity=None, excl
def _interval_query(self, chrom, start, end):
q = Q(xpos__range=(get_xpos(chrom, start), get_xpos(chrom, end)))
- if hasattr(self.model, 'endChrom'):
- q |= Q(endChrom__isnull=True, chrom=chrom, end__range=(start, end))
- q |= Q(endChrom=chrom, end__range=(start, end))
+ if hasattr(self.model, 'end_chrom'):
+ q |= Q(end_chrom__isnull=True, chrom=chrom, end__range=(start, end))
+ q |= Q(end_chrom=chrom, end__range=(start, end))
elif hasattr(self.model, 'end'):
q |= Q(chrom=chrom, end__range=(start, end))
q |= Q(chrom=chrom, pos__lte=start, end__gte=end)
@@ -622,28 +624,22 @@ def has_annotation(self, field):
class EntriesManager(SearchQuerySet):
+ MAX_XPOS_FILTER_INTERVALS = 500
GENOTYPE_LOOKUP = {
- REF_REF: (0,),
- REF_ALT: (1,),
- ALT_ALT: (2,),
- HAS_ALT: (0, '{field} > {value}'),
- HAS_REF: (2, '{field} < {value}'),
+ REF_REF: [0],
+ REF_ALT: [1],
+ ALT_ALT: [2],
+ HAS_ALT: [1, 2],
+ HAS_REF: [0, 1],
+ None: [-1, 0, 1, 2],
}
COMP_HET_ALT = 'COMP_HET_ALT'
GENOTYPE_LOOKUP[COMP_HET_ALT] = GENOTYPE_LOOKUP[REF_ALT]
NULLABLE_GENOTYPE_LOOKUP = {
**GENOTYPE_LOOKUP,
- REF_REF: (0, 'or(isNull({field}), {field} = {value})'),
- HAS_REF: (2, 'or(isNull({field}), {field} < {value})'),
- HAS_ALT: (0, 'and(isNotNull({field}), {field} > {value})'),
- }
- NULLABLE_GENOTYPE_LOOKUP[COMP_HET_ALT] = NULLABLE_GENOTYPE_LOOKUP[HAS_ALT]
- INVALID_NUM_ALT_LOOKUP = {
- (0,): [1, 2],
- (1,): [0, 2],
- (2,): [0, 1],
- (0, '{field} > {value}'): [0],
- (2, '{field} < {value}'): [2],
+ COMP_HET_ALT: GENOTYPE_LOOKUP[HAS_ALT],
+ REF_REF: [-1] + GENOTYPE_LOOKUP[REF_REF],
+ HAS_REF: [-1] + GENOTYPE_LOOKUP[HAS_REF],
}
INHERITANCE_FILTERS = {
@@ -660,7 +656,6 @@ def annotations_model(self):
def call_fields(self):
return dict(self.model.CALL_FIELDS)
-
@property
def genotype_lookup(self):
return self.NULLABLE_GENOTYPE_LOOKUP if self.annotations_model.GENOTYPE_OVERRIDE_FIELDS else self.GENOTYPE_LOOKUP
@@ -689,7 +684,7 @@ def clinvar_model(self):
return self.model.clinvar_join.rel.related_model
def search(self, sample_data, parsed_locus=None, freqs=None, annotations=None, **kwargs):
- entries = self.filter_intervals(**(parsed_locus or {}))
+ entries = self.filter_locus(**(parsed_locus or {}))
entries = self._join_annotations(entries)
@@ -700,7 +695,10 @@ def search(self, sample_data, parsed_locus=None, freqs=None, annotations=None, *
gnomad_filter = (freqs or {}).get('gnomad_genomes') or {}
if hasattr(self.model, 'is_gnomad_gt_5_percent') and ((gnomad_filter.get('af') or 1) <= 0.05 or any(gnomad_filter.get(field) is not None for field in ['ac', 'hh'])):
- entries = entries.filter(is_gnomad_gt_5_percent=False)
+ # Passing field=Value(False) to the django filter causes the SQL to evaluate to "field = false",
+ # while passing field=False evaluates to "NOT field".
+ # For fields used for pruning the table based on the order_by for the table, the former is needed
+ entries = entries.filter(is_gnomad_gt_5_percent=Value(False))
if (annotations or {}).get(NEW_SV_FIELD) and 'newCall' in self.call_fields:
entries = entries.filter(calls__array_exists={'newCall': (None, '{field}')})
@@ -735,16 +733,20 @@ def result_values(self, sample_data):
def _has_clinvar(self):
return hasattr(self.model, 'clinvar_join')
- def _search_call_data(self, entries, sample_data, inheritance_mode=None, inheritance_filter=None, qualityFilter=None, pathogenicity=None, annotate_carriers=False, annotate_hom_alts=False, **kwargs):
- project_guids = {s['project_guid'] for s in sample_data}
- project_filter = Q(project_guid__in=project_guids) if len(project_guids) > 1 else Q(project_guid=sample_data[0]['project_guid'])
+ def _search_call_data(self, entries, sample_data, inheritance_mode=None, inheritance_filter=None, qualityFilter=None, pathogenicity=None, annotate_carriers=False, annotate_hom_alts=False, skip_individual_guid=False, **kwargs):
+ project_guids = sample_data['project_guids']
+ project_filter = Q(project_guid__in=project_guids) if len(project_guids) > 1 else Q(project_guid=project_guids[0])
entries = entries.filter(project_filter)
- sample_type_families, multi_sample_type_families = self._get_family_sample_types(sample_data)
+ multi_sample_type_families = sample_data['sample_type_families'].get('multi', [])
family_q = None
+ multi_sample_type_family_q = None
if multi_sample_type_families:
- family_q = Q(family_guid__in=multi_sample_type_families.keys())
- for sample_type, families in sample_type_families.items():
+ family_q = Q(family_guid__in=multi_sample_type_families)
+ multi_sample_type_family_q = family_q
+ for sample_type, families in sample_data['sample_type_families'].items():
+ if sample_type == 'multi':
+ continue
sample_family_q = Q(family_guid__in=families)
if not self.single_sample_type:
sample_family_q &= Q(sample_type=sample_type)
@@ -755,135 +757,84 @@ def _search_call_data(self, entries, sample_data, inheritance_mode=None, inherit
entries = entries.filter(family_q)
+ inheritance_q = None
+ quality_q = None
+ gt_filter = None
+ family_missing_type_samples = None
quality_filter = qualityFilter or {}
individual_genotype_filter = (inheritance_filter or {}).get('genotype')
- custom_affected = (inheritance_filter or {}).get('affected') or {}
if inheritance_mode or individual_genotype_filter or quality_filter:
clinvar_override_q = AnnotationsQuerySet._clinvar_path_q(
pathogenicity, _get_range_q=lambda path_range: Q(clinvar_join__pathogenicity__range=path_range),
) if self._has_clinvar() else None
- call_q = None
- multi_sample_type_quality_q = None
- multi_sample_type_any_affected_q = None
- for s in sample_data:
- sample_filters, any_affected_samples = self._family_sample_filters(s, inheritance_mode, individual_genotype_filter, quality_filter, custom_affected)
- if not (sample_filters or any_affected_samples):
- continue
- if s['family_guid'] in multi_sample_type_families:
- multi_sample_type_quality_q, multi_sample_type_any_affected_q = self._multi_sample_type_family_calls_q(
- multi_sample_type_quality_q, multi_sample_type_any_affected_q, s, sample_filters, any_affected_samples,
- clinvar_override_q, multi_sample_type_families,
- )
- else:
- sample_type = s['sample_types'][0]
- call_q = self._family_calls_q(
- call_q, s, sample_filters, sample_type, clinvar_override_q,
- any_affected_samples=any_affected_samples, filter_sample_type=not self.single_sample_type
- )
-
- # With families with multiple sample types, can only filter rows after aggregating
- filtered_multi_sample_type_families = {
- family_guid: filters for family_guid, filters in multi_sample_type_families.items() if filters
- }
- if filtered_multi_sample_type_families:
- multi_type_q = Q(family_guid__in=filtered_multi_sample_type_families.keys())
- call_q = (call_q | multi_type_q) if call_q else multi_type_q
- entries = entries.annotate(passes_quality=~multi_type_q | (multi_sample_type_quality_q or Value(True)))
- if multi_sample_type_any_affected_q:
- entries = entries.annotate(passes_inheritance=~multi_type_q | multi_sample_type_any_affected_q)
- else:
- entries = self._annotate_failed_family_samples(entries, filtered_multi_sample_type_families)
-
- if call_q:
- entries = entries.filter(call_q)
-
+ inheritance_q, quality_q, gt_filter, family_missing_type_samples = self._get_inheritance_quality_qs(
+ sample_data, multi_sample_type_families, inheritance_mode, individual_genotype_filter, quality_filter, clinvar_override_q,
+ custom_affected=(inheritance_filter or {}).get('affected') or {},
+ )
if quality_filter.get('vcf_filter'):
q = Q(filters__len=0)
if clinvar_override_q:
q |= clinvar_override_q
entries = entries.filter(q)
- return self._annotate_calls(entries, sample_data, annotate_carriers, annotate_hom_alts, multi_sample_type_families)
+ if multi_sample_type_families:
+ if gt_filter:
+ inheritance_q |= multi_sample_type_family_q
+ entries = self._annotate_failed_family_samples(entries, gt_filter, family_missing_type_samples)
+ elif inheritance_q is not None:
+ entries = entries.annotate(passes_inheritance=inheritance_q)
+ inheritance_q = Q(passes_inheritance=True) | multi_sample_type_family_q
+ else:
+ entries = entries.annotate(passes_inheritance=Value(True))
- @staticmethod
- def _get_family_sample_types(sample_data):
- sample_type_families = defaultdict(list)
- multi_sample_type_families = {}
- for s in sample_data:
- if len(s['sample_types']) == 1:
- sample_type_families[s['sample_types'][0]].append(s['family_guid'])
- else:
- multi_sample_type_families[s['family_guid']] = []
- return sample_type_families, multi_sample_type_families
+ if quality_q is None:
+ entries = entries.annotate(passes_quality=Value(True))
+ else:
+ entries = entries.annotate(passes_quality=quality_q)
+ quality_q = Q(passes_quality=True) | multi_sample_type_family_q
+
+ if inheritance_q is not None:
+ entries = entries.filter(inheritance_q)
+ if quality_q is not None:
+ entries = entries.filter(quality_q)
+
+ return self._annotate_calls(entries, sample_data, annotate_carriers, annotate_hom_alts, skip_individual_guid, multi_sample_type_families)
- def _family_sample_filters(self, family_sample_data, inheritance_mode, individual_genotype_filter, quality_filter, custom_affected):
- sample_filters = []
- any_affected_samples = []
- for sample in family_sample_data['samples']:
+ def _get_inheritance_quality_qs(self, sample_data, multi_sample_type_families, inheritance_mode, individual_genotype_filter, quality_filter, clinvar_override_q, custom_affected):
+ samples_by_gt = defaultdict(list)
+ affected_samples = []
+ family_missing_type_samples = defaultdict(lambda: defaultdict(list))
+ for sample in sample_data['samples']:
affected = custom_affected.get(sample['individual_guid']) or sample['affected']
- sample_inheritance_filter = self._sample_genotype_filter(sample, affected, inheritance_mode, individual_genotype_filter)
- sample_quality_filter = self._sample_quality_filter(affected, quality_filter)
- if sample_inheritance_filter or sample_quality_filter:
- sample_filters.append((sample['sample_ids_by_type'], sample_inheritance_filter, sample_quality_filter))
- if inheritance_mode == ANY_AFFECTED and affected == AFFECTED:
- any_affected_samples.append(sample['sample_ids_by_type'])
- return sample_filters, any_affected_samples
-
- def _family_calls_q(self, call_q, family_sample_data, sample_filters, sample_type, clinvar_override_q, any_affected_samples=None, filter_sample_type=True):
- family_sample_q = None
- if any_affected_samples:
- affected_sample_ids = [f"'{sample_ids_by_type[sample_type]}'" for sample_ids_by_type in any_affected_samples]
- family_sample_q = Q(calls__array_exists={
- 'gt': self.genotype_lookup[HAS_ALT],
- 'sampleId': (', '.join(affected_sample_ids), 'has([{value}], {field})'),
- })
-
- for sample_ids_by_type, sample_inheritance_filter, sample_quality_filter in sample_filters:
- sample_inheritance_filter['sampleId'] = (f"'{sample_ids_by_type[sample_type]}'",)
- sample_q = Q(
- calls__array_exists={**sample_inheritance_filter, **sample_quality_filter},
- family_guid=family_sample_data['family_guid'],
- )
- if filter_sample_type:
- sample_q &= Q(sample_type=sample_type)
- if clinvar_override_q and sample_quality_filter:
- sample_q |= clinvar_override_q & Q(calls__array_exists=sample_inheritance_filter)
+ genotype = self._sample_genotype(sample, affected, inheritance_mode, individual_genotype_filter)
+ if affected == AFFECTED and (inheritance_mode == ANY_AFFECTED or quality_filter.get('affected_only')):
+ affected_samples.append(sample['sample_id'])
+ if (inheritance_mode and inheritance_mode != ANY_AFFECTED) or individual_genotype_filter:
+ for gt in self.genotype_lookup[genotype]:
+ samples_by_gt[gt].append(sample['sample_id'])
+ if sample['family_guid'] in multi_sample_type_families:
+ sample_type = sample['sample_type']
+ missing_type = Sample.SAMPLE_TYPE_WES if sample_type == Sample.SAMPLE_TYPE_WGS else Sample.SAMPLE_TYPE_WGS
+ if not any(s for s in sample_data['samples'] if s['individual_guid'] == sample['individual_guid'] and s['sample_type'] == missing_type):
+ family_missing_type_samples[sample['family_guid']][missing_type].append(sample['sample_id'])
+
+ inheritance_q = None
+ gt_filter = None
+ if inheritance_mode == ANY_AFFECTED:
+ inheritance_q = Q(calls__array_exists={
+ 'gt': (self.genotype_lookup[HAS_ALT], 'has({value}, {field})'),
+ 'sampleId': (affected_samples, 'has({value}, {field})'),
+ })
+ elif samples_by_gt:
+ gt_filter_map = ', '.join([f"{gt}, {samples_by_gt[gt]}" for gt in [-1, 0, 1, 2]])
+ gt_filter = (gt_filter_map, 'has(map({value})[ifNull({field}, -1)], x.sampleId)')
+ inheritance_q = Q(calls__array_all={'gt': gt_filter})
- if family_sample_q is None:
- family_sample_q = sample_q
- else:
- family_sample_q &= sample_q
-
- if family_sample_q and call_q:
- call_q |= family_sample_q
- return call_q or family_sample_q
-
- def _multi_sample_type_family_calls_q(self, call_q, any_affected_q, family_sample_data, sample_filters, any_affected_samples, clinvar_override_q, multi_sample_type_families):
- sample_quality_filters = []
- for sample_ids_by_type, sample_inheritance_filter, sample_quality_filter in sample_filters:
- if sample_inheritance_filter.get('gt'):
- multi_sample_type_families[family_sample_data['family_guid']].append(
- (sample_ids_by_type, sample_inheritance_filter['gt'])
- )
- if sample_quality_filter:
- sample_quality_filters.append((sample_ids_by_type, {}, sample_quality_filter))
- if sample_quality_filters:
- for sample_type in family_sample_data['sample_types']:
- call_q = self._family_calls_q(
- call_q, family_sample_data, sample_quality_filters, sample_type, clinvar_override_q,
- )
- if any_affected_samples:
- multi_sample_type_families[family_sample_data['family_guid']] = True
- for sample_type in family_sample_data['sample_types']:
- any_affected_q = self._family_calls_q(
- any_affected_q, family_sample_data, [], sample_type, clinvar_override_q,
- any_affected_samples= any_affected_samples
- )
+ quality_q = self._quality_q(quality_filter, affected_samples, clinvar_override_q)
- return call_q, any_affected_q
+ return inheritance_q, quality_q, gt_filter, family_missing_type_samples
- def _sample_genotype_filter(self, sample, affected, inheritance_mode, individual_genotype_filter):
- sample_filter = {}
+ def _sample_genotype(self, sample, affected, inheritance_mode, individual_genotype_filter):
genotype = None
if individual_genotype_filter:
genotype = individual_genotype_filter.get(sample['individual_guid'])
@@ -891,14 +842,10 @@ def _sample_genotype_filter(self, sample, affected, inheritance_mode, individual
genotype = self.INHERITANCE_FILTERS.get(inheritance_mode, {}).get(affected)
if (inheritance_mode == X_LINKED_RECESSIVE and affected == UNAFFECTED and sample['sex'] in MALE_SEXES):
genotype = REF_REF
- if genotype:
- sample_filter['gt'] = self.genotype_lookup[genotype]
- return sample_filter
+ return genotype
- def _sample_quality_filter(self, affected, quality_filter):
- sample_filter = {}
- if quality_filter.get('affected_only') and affected != AFFECTED:
- return sample_filter
+ def _quality_q(self, quality_filter, affected_samples, clinvar_override_q):
+ quality_filter_conditions = {}
for field, scale, *filters in self.quality_filters:
filter_key = f'min_{field}'
@@ -907,39 +854,37 @@ def _sample_quality_filter(self, affected, quality_filter):
value = quality_filter.get(filter_key)
if value:
or_filters = ['isNull({field})', '{field} >= {value}'] + filters
- sample_filter[field] = (value / scale, f'or({", ".join(or_filters)})')
+ quality_filter_conditions[field] = (value / scale, f'or({", ".join(or_filters)})')
+
+ if not quality_filter_conditions:
+ return None
- return sample_filter
+ affected_only = quality_filter.get('affected_only')
+ if affected_only:
+ quality_filter_conditions = {'OR': [
+ quality_filter_conditions,
+ {'sampleId': (affected_samples, 'not has({value}, {field})')}
+ ]}
- @classmethod
- def _annotate_failed_family_samples(cls, entries, family_sample_gt_filters):
- gt_map = []
- missing_sample_map = []
- for family_guid, sample_filters in family_sample_gt_filters.items():
- sample_type_filters = defaultdict(list)
- missing_type_samples = defaultdict(list)
- for sample_ids_by_type, gt_filter in sample_filters:
- for sample_type, sample_id in sample_ids_by_type.items():
- sample_type_filters[sample_type].append(f"'{sample_id}', {cls.INVALID_NUM_ALT_LOOKUP[gt_filter]}")
- if len(sample_ids_by_type) == 1:
- missing_type = Sample.SAMPLE_TYPE_WES if sample_type == Sample.SAMPLE_TYPE_WGS else Sample.SAMPLE_TYPE_WGS
- missing_type_samples[missing_type].append(sample_id)
- sample_type_map = [
- f"'{sample_type}', map({', '.join(type_filters)})"
- for sample_type, type_filters in sample_type_filters.items()
- ]
- gt_map.append(f"'{family_guid}', map({', '.join(sample_type_map)})")
- if missing_type_samples:
- missing_type_map = [f"'{sample_type}', {samples}" for sample_type, samples in missing_type_samples.items()]
- missing_sample_map.append(f"'{family_guid}', map({', '.join(missing_type_map)})")
+ quality_q = Q(calls__array_all=quality_filter_conditions)
+ if clinvar_override_q:
+ quality_q |= clinvar_override_q
+
+ return quality_q
+ @staticmethod
+ def _annotate_failed_family_samples(entries, gt_filter, family_missing_type_samples):
entries = entries.annotate(failed_family_samples= ArrayMap(
ArrayFilter('calls', conditions=[{
- 'gt': (', '.join(gt_map), 'has(map({value})[family_guid][sample_type::String][x.sampleId], {field})'),
+ 'gt': (gt_filter[0], f'not {gt_filter[1]}'),
}]),
mapped_expression='tuple(family_guid, x.sampleId)',
))
- if missing_sample_map:
+ if family_missing_type_samples:
+ missing_sample_map = []
+ for family_guid, sample_types in family_missing_type_samples.items():
+ samples = [f"'{sample_type}', {samples}" for sample_type, samples in sample_types.items()]
+ missing_sample_map.append(f"'{family_guid}', map({', '.join(samples)})")
entries = entries.annotate(
missing_family_samples=ArrayMap(
MapLookup('family_guid', Cast('sample_type', models.StringField()), map_values=', '.join(missing_sample_map)),
@@ -948,7 +893,7 @@ def _annotate_failed_family_samples(cls, entries, family_sample_gt_filters):
)
return entries
- def _annotate_calls(self, entries, sample_data=None, annotate_carriers=False, annotate_hom_alts=False, multi_sample_type_families=None):
+ def _annotate_calls(self, entries, sample_data=None, annotate_carriers=False, annotate_hom_alts=False, skip_individual_guid=False, multi_sample_type_families=None):
carriers_expression = self._carriers_expression(sample_data) if annotate_carriers else None
if carriers_expression:
entries = entries.annotate(carriers=carriers_expression)
@@ -969,10 +914,11 @@ def _annotate_calls(self, entries, sample_data=None, annotate_carriers=False, an
fields.append('seqrPop')
if self._has_clinvar():
fields += ['clinvar', 'clinvar_key']
- if multi_sample_type_families or sample_data is None or len(sample_data) > 1:
+ if multi_sample_type_families or sample_data is None or len(set(sample_data['family_guids'])) > 1:
+ genotype_sample_data = None if skip_individual_guid else sample_data
entries = entries.values(*fields).annotate(
familyGuids=ArraySort(ArrayDistinct(GroupArray('family_guid'))),
- **{'genotypes' if sample_data else 'familyGenotypes': GroupArrayArray(self.genotype_expression(sample_data))},
+ **{'genotypes' if genotype_sample_data else 'familyGenotypes': GroupArrayArray(self.genotype_expression(genotype_sample_data))},
**{col: GroupArrayArray(col) for col in genotype_override_annotations}
)
if carriers_expression:
@@ -993,7 +939,7 @@ def _annotate_calls(self, entries, sample_data=None, annotate_carriers=False, an
mapped_expression='x.1',
output_field=models.ArrayField(models.StringField()),
))
- if any((multi_sample_type_families or {}).values()):
+ if multi_sample_type_families:
entries = self._multi_sample_type_filtered_entries(entries)
else:
if carriers_expression:
@@ -1015,14 +961,12 @@ def _annotate_calls(self, entries, sample_data=None, annotate_carriers=False, an
return entries
def genotype_expression(self, sample_data=None):
- sample_map = []
- for data in sample_data or []:
- family_samples = []
- for s in data['samples']:
- family_samples += [
- f"'{sample_id}', '{s['individual_guid']}'" for sample_id in set(s['sample_ids_by_type'].values())
- ]
- sample_map.append(f"'{data['family_guid']}', map({', '.join(family_samples)})")
+ family_samples = defaultdict(list)
+ for s in (sample_data or {}).get('samples', []):
+ family_samples[s['family_guid']].append(f"'{s['sample_id']}', '{s['individual_guid']}'")
+ sample_map = [
+ f"'{family_guid}', map({', '.join(samples)})" for family_guid, samples in family_samples.items()
+ ]
genotype_expressions = list(self.genotype_fields.keys())
output_base_fields = list(self.genotype_fields.values())
output_field_kwargs = {'group_by_key': 'familyGuid'}
@@ -1040,14 +984,10 @@ def genotype_expression(self, sample_data=None):
)
def _carriers_expression(self, sample_data):
- family_carriers = {}
- for family_sample_data in sample_data:
- family_carriers[family_sample_data['family_guid']] = set()
- for s in family_sample_data['samples']:
- if s['affected'] == UNAFFECTED:
- family_carriers[family_sample_data['family_guid']].update([
- f"'{sample_id}'" for sample_id in s['sample_ids_by_type'].values()
- ])
+ family_carriers = defaultdict(set)
+ for s in sample_data['samples']:
+ if s['affected'] == UNAFFECTED:
+ family_carriers[s['family_guid']].add(f"'{s['sample_id']}'")
if not any(family_carriers.values()):
return None
@@ -1101,16 +1041,13 @@ def _family_passes_expression(pass_field):
mapped_expression='x.1', output_field=models.ArrayField(models.StringField()),
)
- def filter_intervals(self, exclude_intervals=False, intervals=None, gene_intervals=None, variant_ids=None, padded_interval=None, **kwargs):
+ def filter_locus(self, exclude_intervals=False, intervals=None, gene_intervals=None, variant_ids=None, padded_interval=None, **kwargs):
entries = self
if variant_ids:
# although technically redundant, the interval query is applied to the entries table before join and reduces the join size,
# while the full variant_id filter is applied to the annotation table after the join
intervals = [(chrom, pos, pos) for chrom, pos, _, _ in variant_ids]
- if not (gene_intervals or intervals):
- return entries
-
if padded_interval:
pos = padded_interval['start']
padding = int((padded_interval['end'] - pos) * padded_interval['padding'])
@@ -1120,19 +1057,34 @@ def filter_intervals(self, exclude_intervals=False, intervals=None, gene_interva
if exclude_intervals:
intervals = None
else:
- chromosomes = {chrom for chrom, _, _ in (gene_intervals or []) + (intervals or [])}
+ chromosomes = {chrom for chrom, _, _ in list((gene_intervals or {}).values()) + (intervals or [])}
intervals = [(chrom, MIN_POS, MAX_POS) for chrom in chromosomes]
- elif gene_intervals:
- intervals = (intervals or []) + gene_intervals
+ gene_intervals = None
+
+ if not (gene_intervals or intervals):
+ return entries
+
+ locus_q = None
+ if gene_intervals:
+ has_entry_genes = hasattr(self.model, 'is_annotated_in_any_gene')
+ if has_entry_genes and not intervals:
+ entries = entries.filter(is_annotated_in_any_gene=Value(True))
+ if (not has_entry_genes) or exclude_intervals or len(gene_intervals) < self.MAX_XPOS_FILTER_INTERVALS:
+ intervals = list((gene_intervals or {}).values()) + (intervals or [])
+ else:
+ locus_q = Q(geneId_ids__bitmap_has_any=list(gene_intervals.keys()))
if intervals:
interval_q = self._interval_query(*intervals[0])
for interval in intervals[1:]:
interval_q |= self._interval_query(*interval)
- filter_func = entries.exclude if exclude_intervals else entries.filter
- entries = filter_func(interval_q)
+ if locus_q is None:
+ locus_q = interval_q
+ else:
+ locus_q |= interval_q
- return entries
+ filter_func = entries.exclude if exclude_intervals else entries.filter
+ return filter_func(locus_q)
@staticmethod
def _interval_query(chrom, start, end):
diff --git a/clickhouse_search/migrations/0001_initial.py b/clickhouse_search/migrations/0001_initial.py
index 05d5815bc7..b733654f63 100644
--- a/clickhouse_search/migrations/0001_initial.py
+++ b/clickhouse_search/migrations/0001_initial.py
@@ -89,14 +89,16 @@ class Migration(migrations.Migration):
('sample_type', clickhouse_backend.models.Enum8Field(choices=[(1, 'WES'), (2, 'WGS')])),
('xpos', clickhouse_search.backend.fields.UInt64FieldDeltaCodecField()),
('is_gnomad_gt_5_percent', clickhouse_backend.models.BoolField()),
+ ('is_annotated_in_any_gene', clickhouse_backend.models.BoolField()),
+ ('geneId_ids', clickhouse_backend.models.ArrayField(base_field=clickhouse_backend.models.UInt32Field())),
('filters', clickhouse_backend.models.ArrayField(base_field=clickhouse_backend.models.StringField(low_cardinality=True))),
('calls', clickhouse_backend.models.ArrayField(base_field=clickhouse_search.backend.fields.NamedTupleField(base_fields=[('sampleId', clickhouse_backend.models.StringField()), ('gt', clickhouse_backend.models.Enum8Field(blank=True, choices=[(0, 'REF'), (1, 'HET'), (2, 'HOM')], null=True)), ('gq', clickhouse_backend.models.UInt8Field(blank=True, null=True)), ('ab', clickhouse_backend.models.DecimalField(blank=True, decimal_places=5, max_digits=9, null=True)), ('dp', clickhouse_backend.models.UInt16Field(blank=True, null=True))]))),
('sign', clickhouse_backend.models.Int8Field()),
],
options={
'db_table': 'GRCh38/SNV_INDEL/entries',
- 'engine': clickhouse_search.backend.engines.CollapsingMergeTree('sign', deduplicate_merge_projection_mode='rebuild', index_granularity=8192, order_by=('project_guid', 'family_guid', 'is_gnomad_gt_5_percent', 'sample_type', 'key'), partition_by='project_guid'),
- 'projection': clickhouse_search.models.Projection('xpos_projection', order_by='xpos, is_gnomad_gt_5_percent'),
+ 'engine': clickhouse_search.backend.engines.CollapsingMergeTree('sign', deduplicate_merge_projection_mode='rebuild', index_granularity=8192, order_by=('project_guid', 'family_guid', 'sample_type', 'is_gnomad_gt_5_percent', 'is_annotated_in_any_gene', 'key'), partition_by='project_guid'),
+ 'projection': clickhouse_search.models.Projection('xpos_projection', order_by='is_gnomad_gt_5_percent, is_annotated_in_any_gene, xpos'),
},
managers=[
('objects', django.db.models.manager.Manager()),
diff --git a/clickhouse_search/migrations/0002_annotationsdiskgrch37snvindel_and_more.py b/clickhouse_search/migrations/0002_annotationsdiskgrch37snvindel_and_more.py
index ef42cbe289..27915f3895 100644
--- a/clickhouse_search/migrations/0002_annotationsdiskgrch37snvindel_and_more.py
+++ b/clickhouse_search/migrations/0002_annotationsdiskgrch37snvindel_and_more.py
@@ -82,6 +82,8 @@ class Migration(migrations.Migration):
('sample_type', clickhouse_backend.models.Enum8Field(choices=[(1, 'WES'), (2, 'WGS')])),
('xpos', clickhouse_search.backend.fields.UInt64FieldDeltaCodecField()),
('is_gnomad_gt_5_percent', clickhouse_backend.models.BoolField()),
+ ('is_annotated_in_any_gene', clickhouse_backend.models.BoolField()),
+ ('geneId_ids', clickhouse_backend.models.ArrayField(base_field=clickhouse_backend.models.UInt32Field())),
('filters', clickhouse_backend.models.ArrayField(base_field=clickhouse_backend.models.StringField(low_cardinality=True))),
('calls', clickhouse_backend.models.ArrayField(base_field=clickhouse_search.backend.fields.NamedTupleField(base_fields=[('sampleId', clickhouse_backend.models.StringField()), ('gt', clickhouse_backend.models.Enum8Field(blank=True, choices=[(0, 'REF'), (1, 'HET'), (2, 'HOM')], null=True)), ('gq', clickhouse_backend.models.UInt8Field(blank=True, null=True)), ('ab', clickhouse_backend.models.DecimalField(blank=True, decimal_places=5, max_digits=9, null=True)), ('dp', clickhouse_backend.models.UInt16Field(blank=True, null=True))]))),
('sign', clickhouse_backend.models.Int8Field()),
@@ -89,8 +91,8 @@ class Migration(migrations.Migration):
options={
'db_table': 'GRCh37/SNV_INDEL/entries',
'abstract': False,
- 'engine': clickhouse_search.backend.engines.CollapsingMergeTree('sign', deduplicate_merge_projection_mode='rebuild', index_granularity=8192, order_by=('project_guid', 'family_guid', 'is_gnomad_gt_5_percent', 'sample_type', 'key'), partition_by='project_guid'),
- 'projection': clickhouse_search.models.Projection('xpos_projection', order_by='xpos, is_gnomad_gt_5_percent'),
+ 'engine': clickhouse_search.backend.engines.CollapsingMergeTree('sign', deduplicate_merge_projection_mode='rebuild', index_granularity=8192, order_by=('project_guid', 'family_guid', 'sample_type', 'is_gnomad_gt_5_percent', 'is_annotated_in_any_gene', 'key'), partition_by='project_guid'),
+ 'projection': clickhouse_search.models.Projection('xpos_projection', order_by='is_gnomad_gt_5_percent, is_annotated_in_any_gene, xpos'),
},
managers=[
('objects', django.db.models.manager.Manager()),
diff --git a/clickhouse_search/migrations/0014_gnomadgenomesgrch37snvindel_gnomadgenomessnvindel.py b/clickhouse_search/migrations/0014_gnomadgenomesgrch37snvindel_gnomadgenomessnvindel.py
new file mode 100644
index 0000000000..5b621c8477
--- /dev/null
+++ b/clickhouse_search/migrations/0014_gnomadgenomesgrch37snvindel_gnomadgenomessnvindel.py
@@ -0,0 +1,85 @@
+# Generated by Django 4.2.21 on 2025-08-13 19:35 and editted manually by the seqr team
+
+import os
+from string import Template
+
+import clickhouse_backend.models
+from django.db import migrations, models
+import django.db.models.deletion
+import django.db.models.manager
+
+CLICKHOUSE_WRITER_PASSWORD = os.environ.get('CLICKHOUSE_WRITER_PASSWORD', 'clickhouse_test')
+CLICKHOUSE_WRITER_USER = os.environ.get('CLICKHOUSE_WRITER_USER', 'clickhouse')
+
+GNOMAD_GENOMES_DICT = Template(Template("""
+CREATE DICTIONARY `$reference_genome/$dataset_type/gnomad_genomes_dict`
+(
+ key UInt32,
+ filter_af Decimal(9, 5)
+)
+PRIMARY KEY key
+SOURCE(CLICKHOUSE(USER $clickhouse_writer_user PASSWORD $clickhouse_writer_password TABLE `$reference_genome/$dataset_type/gnomad_genomes`))
+LIFETIME(MIN 0 MAX 0)
+LAYOUT(FLAT(MAX_ARRAY_SIZE $size))
+""").safe_substitute(
+ # Note the nested Template-ing that allows
+ # double substitution these shared values
+ clickhouse_writer_user=CLICKHOUSE_WRITER_USER,
+ clickhouse_writer_password=CLICKHOUSE_WRITER_PASSWORD,
+))
+
+
+class Migration(migrations.Migration):
+
+ dependencies = [
+ ('clickhouse_search', '0013_annotationsdiskgcnv_annotationsdiskmito_and_more'),
+ ]
+
+ operations = [
+ migrations.CreateModel(
+ name='GnomadGenomesGRCh37SnvIndel',
+ fields=[
+ ('key', models.OneToOneField(db_column='key', on_delete=django.db.models.deletion.CASCADE, primary_key=True, serialize=False, to='clickhouse_search.annotationsgrch37snvindel')),
+ ('filter_af', clickhouse_backend.models.DecimalField(decimal_places=5, max_digits=9)),
+ ],
+ options={
+ 'db_table': 'GRCh37/SNV_INDEL/gnomad_genomes',
+ 'engine': clickhouse_backend.models.ReplacingMergeTree(order_by='key', primary_key='key'),
+ },
+ managers=[
+ ('objects', django.db.models.manager.Manager()),
+ ('_overwrite_base_manager', django.db.models.manager.Manager()),
+ ],
+ ),
+ migrations.CreateModel(
+ name='GnomadGenomesSnvIndel',
+ fields=[
+ ('key', models.OneToOneField(db_column='key', on_delete=django.db.models.deletion.CASCADE, primary_key=True, serialize=False, to='clickhouse_search.annotationssnvindel')),
+ ('filter_af', clickhouse_backend.models.DecimalField(decimal_places=5, max_digits=9)),
+ ],
+ options={
+ 'db_table': 'GRCh38/SNV_INDEL/gnomad_genomes',
+ 'engine': clickhouse_backend.models.ReplacingMergeTree(order_by='key', primary_key='key'),
+ },
+ managers=[
+ ('objects', django.db.models.manager.Manager()),
+ ('_overwrite_base_manager', django.db.models.manager.Manager()),
+ ],
+ ),
+ migrations.RunSQL(
+ GNOMAD_GENOMES_DICT.substitute(
+ reference_genome='GRCh37',
+ dataset_type='SNV_INDEL',
+ size=int(1e8),
+ ),
+ hints={'clickhouse': True},
+ ),
+ migrations.RunSQL(
+ GNOMAD_GENOMES_DICT.substitute(
+ reference_genome='GRCh38',
+ dataset_type='SNV_INDEL',
+ size=int(5e8),
+ ),
+ hints={'clickhouse': True},
+ ),
+ ]
diff --git a/clickhouse_search/migrations/0015_remove_gnomadgenomessnvindel_key_and_more.py b/clickhouse_search/migrations/0015_remove_gnomadgenomessnvindel_key_and_more.py
new file mode 100644
index 0000000000..1c82f6d549
--- /dev/null
+++ b/clickhouse_search/migrations/0015_remove_gnomadgenomessnvindel_key_and_more.py
@@ -0,0 +1,27 @@
+# Generated by Django 4.2.22 on 2025-08-15 19:45
+
+from django.db import migrations
+
+
+class Migration(migrations.Migration):
+
+ dependencies = [
+ ('clickhouse_search', '0014_gnomadgenomesgrch37snvindel_gnomadgenomessnvindel'),
+ ]
+
+ operations = [
+ migrations.RunSQL(
+ 'DROP DICTIONARY `GRCh37/SNV_INDEL/gnomad_genomes_dict`',
+ hints={'clickhouse': True},
+ ),
+ migrations.RunSQL(
+ 'DROP DICTIONARY `GRCh38/SNV_INDEL/gnomad_genomes_dict`',
+ hints={'clickhouse': True},
+ ),
+ migrations.DeleteModel(
+ name='GnomadGenomesGRCh37SnvIndel',
+ ),
+ migrations.DeleteModel(
+ name='GnomadGenomesSnvIndel',
+ ),
+ ]
diff --git a/clickhouse_search/models.py b/clickhouse_search/models.py
index c23171abcf..c7e25c23d3 100644
--- a/clickhouse_search/models.py
+++ b/clickhouse_search/models.py
@@ -588,18 +588,20 @@ class BaseEntriesSnvIndel(BaseEntries):
sample_type = models.Enum8Field(choices=[(1, 'WES'), (2, 'WGS')])
is_gnomad_gt_5_percent = models.BoolField()
+ is_annotated_in_any_gene = models.BoolField()
+ geneId_ids = models.ArrayField(models.UInt32Field())
calls = models.ArrayField(NamedTupleField(CALL_FIELDS))
class Meta:
abstract = True
engine = CollapsingMergeTree(
'sign',
- order_by=('project_guid', 'family_guid', 'is_gnomad_gt_5_percent', 'sample_type', 'key'),
+ order_by=('project_guid', 'family_guid', 'sample_type', 'is_gnomad_gt_5_percent', 'is_annotated_in_any_gene', 'key'),
partition_by='project_guid',
deduplicate_merge_projection_mode='rebuild',
index_granularity=8192,
)
- projection = Projection('xpos_projection', order_by='xpos, is_gnomad_gt_5_percent')
+ projection = Projection('xpos_projection', order_by='is_gnomad_gt_5_percent, is_annotated_in_any_gene, xpos')
class EntriesGRCh37SnvIndel(BaseEntriesSnvIndel):
diff --git a/clickhouse_search/search.py b/clickhouse_search/search.py
index 851d3b1a66..f1e82f99b8 100644
--- a/clickhouse_search/search.py
+++ b/clickhouse_search/search.py
@@ -40,18 +40,25 @@ def get_clickhouse_variants(samples, search, user, previous_search_results, geno
inheritance_mode = search.get('inheritance_mode')
has_comp_het = inheritance_mode in {RECESSIVE, COMPOUND_HET}
for dataset_type, sample_data in sample_data_by_dataset_type.items():
- logger.info(f'Loading {dataset_type} data for {len(sample_data)} families', user)
+ logger.info(f'Loading {dataset_type} data for {len(set(sample_data["family_guids"]))} families', user)
entry_cls = ENTRY_CLASS_MAP[genome_version][dataset_type]
annotations_cls = ANNOTATIONS_CLASS_MAP[genome_version][dataset_type]
- family_guid = sample_data[0]['family_guid']
+ family_guid = sample_data['family_guids'][0]
+ is_multi_project = len(sample_data['project_guids']) > 1
+ skip_individual_guid = is_multi_project and dataset_type == Sample.DATASET_TYPE_VARIANT_CALLS
+ dataset_results = []
if inheritance_mode != COMPOUND_HET:
- result_q = _get_search_results_queryset(entry_cls, annotations_cls, sample_data, **search)
- results += list(result_q[:MAX_VARIANTS + 1])
+ result_q = _get_search_results_queryset(entry_cls, annotations_cls, sample_data, skip_individual_guid=skip_individual_guid, **search)
+ dataset_results += _evaluate_results(result_q)
if has_comp_het:
- result_q = _get_data_type_comp_het_results_queryset(entry_cls, annotations_cls, sample_data, **search)
- results += [list(result[1:]) for result in result_q[:MAX_VARIANTS + 1]]
+ result_q = _get_data_type_comp_het_results_queryset(entry_cls, annotations_cls, sample_data, skip_individual_guid=skip_individual_guid, **search)
+ dataset_results += _evaluate_results(result_q, is_comp_het=True)
+
+ if skip_individual_guid:
+ _add_individual_guids(dataset_results, sample_data)
+ results += dataset_results
if has_comp_het and Sample.DATASET_TYPE_VARIANT_CALLS in sample_data_by_dataset_type and any(
dataset_type.startswith(Sample.DATASET_TYPE_SV_CALLS) for dataset_type in sample_data_by_dataset_type
@@ -72,6 +79,13 @@ def _get_search_results_queryset(entry_cls, annotations_cls, sample_data, **sear
return results.result_values()
+def _evaluate_results(result_q, is_comp_het=False):
+ results = [list(result[1:]) if is_comp_het else result for result in result_q[:MAX_VARIANTS + 1]]
+ if len(results) > MAX_VARIANTS:
+ from seqr.utils.search.utils import InvalidSearchException
+ raise InvalidSearchException('This search returned too many results')
+ return results
+
def _get_multi_data_type_comp_het_results_queryset(genome_version, sample_data_by_dataset_type, annotations=None, annotations_secondary=None, inheritance_mode=None, **search_kwargs):
if annotations_secondary:
annotations = {
@@ -81,28 +95,42 @@ def _get_multi_data_type_comp_het_results_queryset(genome_version, sample_data_b
entry_cls = ENTRY_CLASS_MAP[genome_version][Sample.DATASET_TYPE_VARIANT_CALLS]
annotations_cls = ANNOTATIONS_CLASS_MAP[genome_version][Sample.DATASET_TYPE_VARIANT_CALLS]
- snv_indel_families = {s['family_guid'] for s in sample_data_by_dataset_type[Sample.DATASET_TYPE_VARIANT_CALLS]}
+ snv_indel_sample_data = sample_data_by_dataset_type[Sample.DATASET_TYPE_VARIANT_CALLS]
+ snv_indel_families = set(snv_indel_sample_data['family_guids'])
+ skip_individual_guid = len(snv_indel_sample_data['project_guids']) > 1
results = []
for sample_type in [Sample.SAMPLE_TYPE_WES, Sample.SAMPLE_TYPE_WGS]:
sv_dataset_type = f'{Sample.DATASET_TYPE_SV_CALLS}_{sample_type}'
- sv_families = {s['family_guid'] for s in sample_data_by_dataset_type.get(sv_dataset_type, [])}
+ sv_sample_data = sample_data_by_dataset_type.get(sv_dataset_type, {})
+ sv_families = set(sv_sample_data.get('family_guids', []))
families = snv_indel_families.intersection(sv_families)
if not families:
continue
- entries = entry_cls.objects.search([
- s for s in sample_data_by_dataset_type[Sample.DATASET_TYPE_VARIANT_CALLS] if s['family_guid'] in families
- ], **search_kwargs, annotations=annotations, inheritance_mode=COMPOUND_HET_ALLOW_HOM_ALTS, annotate_carriers=True, annotate_hom_alts=True)
+
+ entries = entry_cls.objects.search({
+ **snv_indel_sample_data,
+ 'family_guids': list(families),
+ 'samples': [s for s in snv_indel_sample_data['samples'] if s['family_guid'] in families]
+ }, skip_individual_guid=skip_individual_guid, **search_kwargs, annotations=annotations, inheritance_mode=COMPOUND_HET_ALLOW_HOM_ALTS, annotate_carriers=True, annotate_hom_alts=True)
snv_indel_q = annotations_cls.objects.subquery_join(entries).search(**search_kwargs, annotations=annotations)
- sv_entries = ENTRY_CLASS_MAP[genome_version][sv_dataset_type].objects.search([
- s for s in sample_data_by_dataset_type[sv_dataset_type] if s['family_guid'] in families
- ], **search_kwargs, annotations=annotations, inheritance_mode=COMPOUND_HET, annotate_carriers=True)
+ sv_sample_data = {
+ **sv_sample_data,
+ 'family_guids': list(families),
+ 'samples': [s for s in sv_sample_data['samples'] if s['family_guid'] in families]
+ }
+ sv_entries = ENTRY_CLASS_MAP[genome_version][sv_dataset_type].objects.search(
+ sv_sample_data, **search_kwargs, annotations=annotations, inheritance_mode=COMPOUND_HET, annotate_carriers=True,
+ )
sv_annotations_cls = ANNOTATIONS_CLASS_MAP[genome_version][sv_dataset_type]
sv_q = sv_annotations_cls.objects.subquery_join(sv_entries).search(**search_kwargs, annotations=annotations)
result_q = _get_comp_het_results_queryset(annotations_cls, snv_indel_q, sv_q, len(families))
- results += [list(result[1:]) for result in result_q[:MAX_VARIANTS + 1]]
+ dataset_results = _evaluate_results(result_q, is_comp_het=True)
+ if skip_individual_guid:
+ _add_individual_guids(dataset_results, sv_sample_data)
+ results += dataset_results
return results
@@ -117,7 +145,7 @@ def _get_data_type_comp_het_results_queryset(entry_cls, annotations_cls, sample_
annotations=annotations_secondary or annotations, **search_kwargs,
)
- return _get_comp_het_results_queryset(annotations_cls, primary_q, secondary_q, len(sample_data))
+ return _get_comp_het_results_queryset(annotations_cls, primary_q, secondary_q, len(set(sample_data['family_guids'])))
def _get_comp_het_results_queryset(annotations_cls, primary_q, secondary_q, num_families):
@@ -143,22 +171,50 @@ def _get_comp_het_results_queryset(annotations_cls, primary_q, secondary_q, num_
is_overlapped_del,
F('primary_familyGuids'),
ArrayIntersect('primary_familyGuids', 'primary_no_hom_alt_families'),
- condition='',
+ condition='', output_field=ArrayField(StringField()),
))
if num_families > 1:
+ primary_family_expr = 'primary_familyGuids' if results.has_annotation('primary_familyGuids') else ArrayMap(
+ 'primary_familyGenotypes', mapped_expression='x.1',
+ )
+ secondary_family_expr = 'secondary_familyGuids' if results.has_annotation('secondary_familyGuids') else ArrayMap(
+ 'secondary_familyGenotypes', mapped_expression='x.1',
+ )
+ if results.has_annotation('primary_genotypes'):
+ genotype_expressions = {
+ 'primary_genotypes': ArrayFilter('primary_genotypes', conditions=[
+ {2: ('arrayIntersect(primary_familyGuids, secondary_familyGuids)', 'has({value}, {field})')},
+ ]),
+ 'secondary_genotypes': ArrayFilter('secondary_genotypes', conditions=[
+ {2: ('arrayIntersect(primary_familyGuids, secondary_familyGuids)', 'has({value}, {field})')},
+ ]),
+ }
+ elif results.has_annotation('secondary_genotypes'):
+ genotype_expressions = {
+ 'primary_familyGenotypes': ArrayFilter('primary_familyGenotypes', conditions=[
+ {1: ('secondary_familyGuids', 'has({value}, {field})')},
+ ]),
+ 'secondary_genotypes': ArrayFilter('secondary_genotypes', conditions=[
+ {2: (None, 'arrayExists(g -> g.1 = {field}, primary_familyGenotypes)')},
+ ]),
+ }
+ else:
+ genotype_expressions = {
+ 'primary_familyGenotypes': ArrayFilter('primary_familyGenotypes', conditions=[
+ {1: (None, 'arrayExists(g -> g.1 = {field}, secondary_familyGenotypes)')},
+ ]),
+ 'secondary_familyGenotypes': ArrayFilter('secondary_familyGenotypes', conditions=[
+ {1: (None, 'arrayExists(g -> g.1 = {field}, primary_familyGenotypes)')},
+ ]),
+ }
results = results.annotate(
primary_familyGuids=ArrayIntersect(
- 'primary_familyGuids', 'secondary_familyGuids', output_field=ArrayField(StringField()),
+ primary_family_expr, secondary_family_expr, output_field=ArrayField(StringField()),
),
).filter(primary_familyGuids__not_empty=True).annotate(
secondary_familyGuids=F('primary_familyGuids'),
- primary_genotypes=ArrayFilter('primary_genotypes', conditions=[
- {2: ('arrayIntersect(primary_familyGuids, secondary_familyGuids)', 'has({value}, {field})')},
- ]),
- secondary_genotypes=ArrayFilter('secondary_genotypes', conditions=[
- {2: ('arrayIntersect(primary_familyGuids, secondary_familyGuids)', 'has({value}, {field})')},
- ]),
+ **genotype_expressions,
)
return results.annotate(
@@ -177,6 +233,28 @@ def _result_as_tuple(results, field_prefix):
return Tuple(*fields.keys(), output_field=NamedTupleField(list(fields.values())))
+def _add_individual_guids(results, sample_data):
+ sample_map = {(s['family_guid'], s['sample_id']): s['individual_guid'] for s in sample_data['samples']}
+ for result in results:
+ if isinstance(result, list):
+ for variant in result:
+ _set_individual_guids(variant, sample_map)
+ else:
+ _set_individual_guids(result, sample_map)
+
+
+def _set_individual_guids(result, sample_map):
+ if 'familyGenotypes' not in result:
+ return
+ result['familyGuids'] = sorted(result['familyGenotypes'].keys())
+ individual_genotypes = defaultdict(list)
+ for family_guid, genotypes in result.pop('familyGenotypes').items():
+ for genotype in genotypes:
+ individual_guid = sample_map[(family_guid, genotype['sampleId'])]
+ individual_genotypes[individual_guid].append({**genotype, 'individualGuid': individual_guid})
+ result['genotypes'] = {k: v[0] if len(v) == 1 else v for k, v in individual_genotypes.items()}
+
+
def get_clickhouse_cache_results(results, sort, family_guid):
sort_metadata = _get_sort_gene_metadata(sort, results, family_guid)
sort_key = _get_sort_key(sort, sort_metadata)
@@ -248,40 +326,50 @@ def _is_matched_minimal_transcript(transcript, minimal_transcript):
and transcript.get('spliceregion', {}).get('extended_intronic_splice_region_variant') == minimal_transcript.get('extendedIntronicSpliceRegionVariant'))
-
def _get_sample_data(samples):
+ mismatch_affected_samples = samples.values('sample_id', 'dataset_type').annotate(
+ projects=ArrayAgg('individual__family__project__name', distinct=True),
+ affected=ArrayAgg('individual__affected', distinct=True),
+ ).filter(affected__len__gt=1)
+ if mismatch_affected_samples:
+ from seqr.utils.search.utils import InvalidSearchException
+ raise InvalidSearchException(
+ 'The following samples are incorrectly configured and have different affected statuses in different projects: ' +
+ ', '.join([f'{agg["sample_id"]} ({"/ ".join(agg["projects"])})' for agg in mismatch_affected_samples]),
+ )
+
sample_data = samples.values(
- 'dataset_type', family_guid=F('individual__family__guid'), project_guid=F('individual__family__project__guid'),
+ 'dataset_type', 'sample_type',
).annotate(
- samples=ArrayAgg(JSONObject(affected='individual__affected', sex='individual__sex', sample_id='sample_id', sample_type='sample_type', individual_guid=F('individual__guid'))),
- sample_types=ArrayAgg('sample_type', distinct=True),
+ project_guids=ArrayAgg('individual__family__project__guid', distinct=True),
+ family_guids=ArrayAgg('individual__family__guid', distinct=True),
+ samples=ArrayAgg(JSONObject(affected='individual__affected', sex='individual__sex', sample_id='sample_id', sample_type='sample_type', family_guid=F('individual__family__guid'), individual_guid=F('individual__guid'))),
)
- samples_by_dataset_type = defaultdict(list)
+ samples_by_dataset_type = {}
for data in sample_data:
- samples = _group_by_sample_type(data['samples'])
- if data['dataset_type'] == Sample.DATASET_TYPE_SV_CALLS:
- samples_by_type = defaultdict(list)
- for sample in samples:
- for sample_type in sample['sample_ids_by_type']:
- samples_by_type[sample_type].append(
- {**sample, 'sample_ids_by_type': {sample_type: sample['sample_ids_by_type'][sample_type]}}
- )
- for sample_type, type_samples in samples_by_type.items():
- samples_by_dataset_type[f"{data['dataset_type']}_{sample_type}"].append({**data, 'samples': type_samples, 'sample_types': [sample_type]})
+ dataset_type = data.pop('dataset_type')
+ sample_type = data.pop('sample_type')
+ if dataset_type == Sample.DATASET_TYPE_SV_CALLS:
+ dataset_type = f'{Sample.DATASET_TYPE_SV_CALLS}_{sample_type}'
+
+ if dataset_type in samples_by_dataset_type:
+ other_type_data = samples_by_dataset_type[dataset_type]
+ other_sample_type = next(iter(other_type_data['sample_type_families'].keys()))
+ family_guids = set(data['family_guids'])
+ other_type_family_guids = set(other_type_data['family_guids'])
+ sample_type_families = {
+ other_sample_type: other_type_family_guids - family_guids,
+ sample_type: family_guids - other_type_family_guids,
+ 'multi': family_guids.intersection(other_type_family_guids),
+ }
+ data['sample_type_families'] = {k: v for k, v in sample_type_families.items() if v}
+ for key in ['project_guids', 'family_guids', 'samples']:
+ data[key] += other_type_data[key]
else:
- samples_by_dataset_type[data['dataset_type']].append({**data, 'samples': samples})
- return samples_by_dataset_type
-
+ data['sample_type_families'] = {sample_type: set(data['family_guids'])}
-def _group_by_sample_type(samples):
- samples_by_individual_type = {}
- for sample in samples:
- sample_type = sample.pop('sample_type')
- sample_id = sample.pop('sample_id')
- if sample['individual_guid'] not in samples_by_individual_type:
- samples_by_individual_type[sample['individual_guid']] = {'sample_ids_by_type': {}, **sample}
- samples_by_individual_type[sample['individual_guid']]['sample_ids_by_type'][sample_type] = sample_id
- return list(samples_by_individual_type.values())
+ samples_by_dataset_type[dataset_type] = data
+ return samples_by_dataset_type
OMIM_SORT = 'in_omim'
@@ -429,7 +517,7 @@ def _clickhouse_variant_lookup(variant_id, genome_version, data_type, samples):
keys = KEY_LOOKUP_CLASS_MAP[genome_version][data_type].objects.filter(variant_id=variant_id).values_list('key', flat=True)
entries = entry_cls.objects.filter(key__in=keys)
else:
- entries = entry_cls.objects.filter_intervals(variant_ids=[variant_id])
+ entries = entry_cls.objects.filter_locus(variant_ids=[variant_id])
entries = entries.result_values(sample_data)
results = annotations_cls.objects.subquery_join(entries)
@@ -464,7 +552,7 @@ def _add_liftover_genotypes(variant, data_type, variant_id):
).values_list('key', flat=True)
if not keys:
return
- lifted_entries = lifted_entry_cls.objects.filter_intervals(variant_ids=[variant_id]).filter(key=keys[0])
+ lifted_entries = lifted_entry_cls.objects.filter_locus(variant_ids=[variant_id]).filter(key=keys[0])
lifted_entry_data = lifted_entries.values('key').annotate(
familyGenotypes=GroupArrayArray(lifted_entry_cls.objects.genotype_expression())
)
@@ -538,9 +626,10 @@ def delete_clickhouse_project(project, dataset_type=None, **kwargs):
table_base = f'{GENOME_VERSION_LOOKUP[project.genome_version]}/{dataset_type}'
with connections['clickhouse_write'].cursor() as cursor:
cursor.execute(f'ALTER TABLE "{table_base}/entries" DROP PARTITION %s', [project.guid])
- cursor.execute(f'ALTER TABLE "{table_base}/project_gt_stats" DROP PARTITION %s', [project.guid])
- view_name = f'{table_base}/project_gt_stats_to_gt_stats_mv'
- cursor.execute(f'SYSTEM REFRESH VIEW "{view_name}"')
- cursor.execute(f'SYSTEM WAIT VIEW "{view_name}"')
- cursor.execute(f'SYSTEM RELOAD DICTIONARY "{table_base}/gt_stats_dict"')
+ if dataset_type != 'GCNV':
+ cursor.execute(f'ALTER TABLE "{table_base}/project_gt_stats" DROP PARTITION %s', [project.guid])
+ view_name = f'{table_base}/project_gt_stats_to_gt_stats_mv'
+ cursor.execute(f'SYSTEM REFRESH VIEW "{view_name}"')
+ cursor.execute(f'SYSTEM WAIT VIEW "{view_name}"')
+ cursor.execute(f'SYSTEM RELOAD DICTIONARY "{table_base}/gt_stats_dict"')
return f'Deleted all {dataset_type} search data for project {project.name}'
diff --git a/clickhouse_search/search_tests.py b/clickhouse_search/search_tests.py
index d3c466f762..2d5a8b5f93 100644
--- a/clickhouse_search/search_tests.py
+++ b/clickhouse_search/search_tests.py
@@ -217,7 +217,7 @@ def test_multi_project_search(self):
)
def test_both_sample_types_search(self):
- Sample.objects.exclude(dataset_type='SNV_INDEL').update(is_active=False)
+ Sample.objects.filter(dataset_type='MITO').update(is_active=False)
# One family (F000011_11) in a multi-project search has identical exome and genome data.
self._set_multi_project_search()
@@ -235,20 +235,48 @@ def test_both_sample_types_search(self):
# Variant 3 is inherited in both sample types.
# Variant 4 is de novo in exome, but inherited in genome in the same parent that has variant 3.
self._assert_expected_search(
- [VARIANT1_BOTH_SAMPLE_TYPES, VARIANT4_BOTH_SAMPLE_TYPES],
- inheritance_mode='de_novo', locus=None,
+ [VARIANT1_BOTH_SAMPLE_TYPES, VARIANT2_BOTH_SAMPLE_TYPES, VARIANT3_BOTH_SAMPLE_TYPES, VARIANT4_BOTH_SAMPLE_TYPES,
+ GCNV_VARIANT1, GCNV_VARIANT2, GCNV_VARIANT3, GCNV_VARIANT4],
+ inheritance_mode='any_affected', locus=None,
+ )
+
+ self._assert_expected_search(
+ [VARIANT2_BOTH_SAMPLE_TYPES, VARIANT3_BOTH_SAMPLE_TYPES, GCNV_VARIANT1, GCNV_VARIANT2, GCNV_VARIANT4],
+ inheritance_mode='any_affected', quality_filter={'min_gq': 40, 'min_qs': 20},
+ )
+
+ self._assert_expected_search(
+ [VARIANT1_BOTH_SAMPLE_TYPES, VARIANT4_BOTH_SAMPLE_TYPES, GCNV_VARIANT1],
+ inheritance_mode='de_novo', quality_filter=None,
)
self._add_sample_type_samples('WGS', guid__in=['S000133_hg00732', 'S000134_hg00733'])
self._assert_expected_search(
- [VARIANT1_BOTH_SAMPLE_TYPES, VARIANT2_BOTH_SAMPLE_TYPES, VARIANT4_BOTH_SAMPLE_TYPES],
+ [VARIANT1_BOTH_SAMPLE_TYPES, VARIANT2_BOTH_SAMPLE_TYPES, VARIANT3_BOTH_SAMPLE_TYPES,
+ VARIANT4_BOTH_SAMPLE_TYPES, GCNV_VARIANT1, GCNV_VARIANT2, GCNV_VARIANT3, GCNV_VARIANT4],
+ inheritance_mode='any_affected',
+ )
+
+ self._assert_expected_search(
+ [VARIANT1_BOTH_SAMPLE_TYPES, VARIANT2_BOTH_SAMPLE_TYPES, VARIANT4_BOTH_SAMPLE_TYPES, GCNV_VARIANT1],
inheritance_mode='de_novo',
)
self._assert_expected_search(
- [VARIANT1_BOTH_SAMPLE_TYPES, VARIANT2_BOTH_SAMPLE_TYPES, [VARIANT3_BOTH_SAMPLE_TYPES, VARIANT4_BOTH_SAMPLE_TYPES]],
- inheritance_mode='recessive', **COMP_HET_ALL_PASS_FILTERS, cached_variant_fields=[
- {}, {}, [{'selectedGeneId': 'ENSG00000097046'}, {'selectedGeneId': 'ENSG00000097046'}],
+ [VARIANT2_BOTH_SAMPLE_TYPES, GCNV_VARIANT1],
+ inheritance_mode='de_novo', quality_filter={'min_gq': 40}
+ )
+
+ self.maxDiff = None
+ self._assert_expected_search(
+ [VARIANT1_BOTH_SAMPLE_TYPES, VARIANT2_BOTH_SAMPLE_TYPES,
+ [{**VARIANT2_BOTH_SAMPLE_TYPES, 'selectedMainTranscriptId': 'ENST00000450625'}, GCNV_VARIANT4],
+ [VARIANT3_BOTH_SAMPLE_TYPES, VARIANT4_BOTH_SAMPLE_TYPES],
+ GCNV_VARIANT3, [GCNV_VARIANT3, GCNV_VARIANT4]],
+ inheritance_mode='recessive', **COMP_HET_ALL_PASS_FILTERS, quality_filter=None, cached_variant_fields=[
+ {}, {}, [{'selectedGeneId': 'ENSG00000277258'}, {'selectedGeneId': 'ENSG00000277258'}],
+ [{'selectedGeneId': 'ENSG00000097046'}, {'selectedGeneId': 'ENSG00000097046'}], {},
+ [{'selectedGeneId': 'ENSG00000275023'}, {'selectedGeneId': 'ENSG00000275023'}],
]
)
@@ -615,6 +643,22 @@ def test_variant_id_search(self):
[],locus={'rawVariantItems': VARIANT_IDS[1]},
)
+ @mock.patch('clickhouse_search.search.MAX_VARIANTS', 3)
+ def test_invalid_search(self):
+ with self.assertRaises(InvalidSearchException) as cm:
+ self._assert_expected_search([])
+ self.assertEqual(str(cm.exception),'This search returned too many results')
+
+ Sample.objects.filter(guid='S000143_na20885').update(sample_id='HG00732')
+ self._set_multi_project_search()
+ with self.assertRaises(InvalidSearchException) as cm:
+ self._assert_expected_search([], locus={'rawItems': GENE_IDS[0]})
+ self.assertEqual(
+ str(cm.exception),
+ 'The following samples are incorrectly configured and have different affected statuses in different projects: '
+ 'HG00732 (1kg project nåme with uniçøde/ Test Reprocessed Project)',
+ )
+
def test_variant_lookup(self):
variant = variant_lookup(self.user, ('1', 10439, 'AC', 'A'))
self._assert_expected_variants([variant], [VARIANT_LOOKUP_VARIANT])
@@ -643,8 +687,9 @@ def test_variant_lookup(self):
variants = sv_variant_lookup(self.user, 'phase2_DEL_chr14_4640', families, sample_type='WGS')
self._assert_expected_variants(variants, [SV_VARIANT4, GCNV_VARIANT4])
+ # reciprocal overlap does not meet the threshold for smaller events
variants = sv_variant_lookup(self.user, 'suffix_140608_DUP', families, sample_type='WES')
- self._assert_expected_variants(variants, [GCNV_VARIANT4, SV_VARIANT4])
+ self._assert_expected_variants(variants, [GCNV_VARIANT4])
variants = sv_variant_lookup(self.user, 'suffix_140593_DUP', families, sample_type='WES')
self._assert_expected_variants(variants, [GCNV_VARIANT3])
@@ -687,7 +732,7 @@ def test_frequency_filter(self):
self._assert_expected_search(
[MULTI_FAMILY_VARIANT, VARIANT4, GCNV_VARIANT2, GCNV_VARIANT3, GCNV_VARIANT4, MITO_VARIANT1, MITO_VARIANT2, MITO_VARIANT3],
- freqs={'callset': {'ac': 6}, **sv_callset_filter},
+ freqs={'callset': {'ac': 7}, **sv_callset_filter},
)
self._assert_expected_search(
@@ -695,7 +740,7 @@ def test_frequency_filter(self):
)
self._assert_expected_search(
- [MULTI_FAMILY_VARIANT, GCNV_VARIANT3, MITO_VARIANT1, MITO_VARIANT2, MITO_VARIANT3], freqs={'callset': {'ac': 6, 'hh': 0}, 'sv_callset': {'ac': 50}},
+ [MULTI_FAMILY_VARIANT, GCNV_VARIANT3, MITO_VARIANT1, MITO_VARIANT2, MITO_VARIANT3], freqs={'callset': {'ac': 7, 'hh': 0}, 'sv_callset': {'ac': 50}},
)
self._set_sv_family_search()
@@ -712,10 +757,18 @@ def test_frequency_filter(self):
[VARIANT2, VARIANT4, GCNV_VARIANT1, GCNV_VARIANT2, GCNV_VARIANT3, GCNV_VARIANT4, MITO_VARIANT1, MITO_VARIANT2], freqs={'gnomad_genomes': {'af': 0.05, 'hh': 1}, 'gnomad_mito': {'af': 0.05}},
)
+ self._assert_expected_search(
+ [VARIANT4, GCNV_VARIANT3, MITO_VARIANT1, MITO_VARIANT2, MITO_VARIANT3],
+ freqs={'topmed': {'af': 0.05, 'hh': 1}, 'sv_callset': {'ac': 50}},
+ )
+
self._set_sv_family_search()
self._assert_expected_search(
[SV_VARIANT1, SV_VARIANT3, SV_VARIANT4], freqs={'gnomad_svs': {'af': 0.001}},
)
+ self._assert_expected_search(
+ [SV_VARIANT1, SV_VARIANT3, SV_VARIANT4], freqs={'gnomad_svs': {'ac': 4000}},
+ )
self._reset_search_families()
self._assert_expected_search(
@@ -742,8 +795,9 @@ def test_frequency_filter(self):
def test_annotations_filter(self):
self._assert_expected_search([VARIANT2], pathogenicity={'hgmd': ['hgmd_other']})
+ self._assert_expected_search([], pathogenicity={'hgmd': ['disease_causing', 'likely_disease_causing']})
- pathogenicity = {'clinvar': ['likely_pathogenic', 'vus_or_conflicting', 'benign']}
+ pathogenicity = {'clinvar': ['likely_pathogenic', 'vus_or_conflicting', 'benign'], 'hgmd': []}
self._assert_expected_search(
[VARIANT1, VARIANT2, MITO_VARIANT1, MITO_VARIANT3], pathogenicity=pathogenicity,
)
@@ -1076,7 +1130,7 @@ def test_secondary_annotations_filter(self):
)
self._add_sample_type_samples('WES', individual__family__guid='F000014_14')
- self.results_model.families.set(Family.objects.filter(guid__in=['F000002_2', 'F000014_14']))
+ self.results_model.families.set(Family.objects.filter(guid__in=['F000002_2', 'F000011_11', 'F000014_14']))
self._assert_expected_search(
[MULTI_DATA_TYPE_COMP_HET_VARIANT2, [MULTI_DATA_TYPE_COMP_HET_VARIANT2, GCNV_VARIANT4], MULTI_PROJECT_GCNV_VARIANT3, [GCNV_VARIANT3, GCNV_VARIANT4]],
inheritance_mode='recessive',
@@ -1091,11 +1145,19 @@ def test_secondary_annotations_filter(self):
# Search works with a different number of samples within the family
self._reset_search_families()
missing_gt_gcnv_variant = {
- **GCNV_VARIANT4, 'genotypes': {k: v for k, v in GCNV_VARIANT4['genotypes'].items() if k != 'I000005_hg00732'}
+ **GCNV_VARIANT4,
+ 'familyGuids': ['F000002_2_x'],
+ 'genotypes': {k: {**v, 'familyGuid': 'F000002_2_x'} for k, v in GCNV_VARIANT4['genotypes'].items() if k != 'I000005_hg00732'}
+ }
+ missing_gt_comp_het_variant = {
+ **MULTI_DATA_TYPE_COMP_HET_VARIANT2,
+ 'familyGuids': ['F000002_2_x'],
+ 'genotypes': {k: {**v, 'familyGuid': 'F000002_2_x'} for k, v in MULTI_DATA_TYPE_COMP_HET_VARIANT2['genotypes'].items()}
}
Sample.objects.filter(guid='S000146_hg00732').update(is_active=False)
+ Family.objects.filter(guid='F000002_2').update(guid='F000002_2_x')
self._assert_expected_search(
- [[MULTI_DATA_TYPE_COMP_HET_VARIANT2, missing_gt_gcnv_variant]],
+ [[missing_gt_comp_het_variant, missing_gt_gcnv_variant]],
inheritance_mode='compound_het', pathogenicity=pathogenicity, locus=None,
annotations=gcnv_annotations_2, annotations_secondary=selected_transcript_annotations, cached_variant_fields=[[
{'selectedGeneId': 'ENSG00000277258', 'selectedTranscript': None},
@@ -1180,7 +1242,7 @@ def test_sort(self):
)
self._assert_expected_search(
- [MITO_VARIANT1, MITO_VARIANT2, MITO_VARIANT3, VARIANT4, MULTI_FAMILY_VARIANT, VARIANT2, VARIANT1, GCNV_VARIANT3, GCNV_VARIANT4, GCNV_VARIANT2, GCNV_VARIANT1],
+ [MITO_VARIANT1, MITO_VARIANT2, MITO_VARIANT3, VARIANT4, MULTI_FAMILY_VARIANT, VARIANT1, VARIANT2, GCNV_VARIANT3, GCNV_VARIANT4, GCNV_VARIANT2, GCNV_VARIANT1],
sort='callset_af',
)
diff --git a/clickhouse_search/test_utils.py b/clickhouse_search/test_utils.py
index 2f3758e368..0395005abf 100644
--- a/clickhouse_search/test_utils.py
+++ b/clickhouse_search/test_utils.py
@@ -21,9 +21,9 @@
)
VARIANT1 = {**deepcopy(HAIL_VARIANT1), 'key': 1, 'populations': {**deepcopy(HAIL_VARIANT1)['populations'], 'seqr': {'ac': 8, 'hom': 3}}}
-VARIANT2 = {**deepcopy(HAIL_VARIANT2), 'key': 2, 'populations': {**deepcopy(HAIL_VARIANT2)['populations'], 'seqr': {'ac': 7, 'hom': 2}}}
-VARIANT3 = {**deepcopy(HAIL_VARIANT3), 'key': 3, 'populations': {**deepcopy(HAIL_VARIANT3)['populations'], 'seqr': {'ac': 6, 'hom': 0}}}
-VARIANT4 = {**deepcopy(HAIL_VARIANT4), 'key': 4, 'populations': {**deepcopy(HAIL_VARIANT4)['populations'], 'seqr': {'ac': 4, 'hom': 1}}}
+VARIANT2 = {**deepcopy(HAIL_VARIANT2), 'key': 2, 'populations': {**deepcopy(HAIL_VARIANT2)['populations'], 'seqr': {'ac': 10, 'hom': 3}}}
+VARIANT3 = {**deepcopy(HAIL_VARIANT3), 'key': 3, 'populations': {**deepcopy(HAIL_VARIANT3)['populations'], 'seqr': {'ac': 7, 'hom': 0}}}
+VARIANT4 = {**deepcopy(HAIL_VARIANT4), 'key': 4, 'populations': {**deepcopy(HAIL_VARIANT4)['populations'], 'seqr': {'ac': 5, 'hom': 1}}}
PROJECT_2_VARIANT = {**deepcopy(HAIL_PROJECT_2_VARIANT), 'key': 5, 'populations': {**deepcopy(HAIL_PROJECT_2_VARIANT)['populations'], 'seqr': {'ac': 2, 'hom': 0}}}
MITO_VARIANT1 = {**deepcopy(HAIL_MITO_VARIANT1), 'key': 6, 'populations': {**deepcopy(HAIL_MITO_VARIANT1)['populations'], 'seqr': {'ac': 0}, 'seqr_heteroplasmy': {'ac': 1}}}
MITO_VARIANT2 = {**deepcopy(HAIL_MITO_VARIANT2), 'key': 7, 'populations': {**deepcopy(HAIL_MITO_VARIANT2)['populations'], 'seqr': {'ac': 0}, 'seqr_heteroplasmy': {'ac': 1}}}
diff --git a/matchmaker/views/external_api_tests.py b/matchmaker/views/external_api_tests.py
index f486a8e091..d1117ce86e 100644
--- a/matchmaker/views/external_api_tests.py
+++ b/matchmaker/views/external_api_tests.py
@@ -253,7 +253,7 @@ def test_mme_match_proxy(self, mock_post_to_slack, mock_email, mock_logger):
be found found at https://seqr.broadinstitute.org/matchmaker/disclaimer."""
match1 = 'seqr ID NA19675_1 from project 1kg project n\u00e5me with uni\u00e7\u00f8de in family 1 inserted into matchbox on May 23, 2018, with seqr link /project/R0001_1kg/family_page/F000001_1/matchmaker_exchange'
match2 = 'seqr ID NA20888 from project Test Reprocessed Project in family 12 inserted into matchbox on Feb 05, 2019, with seqr link /project/R0003_test/family_page/F000012_12/matchmaker_exchange'
- match3 = 'seqr ID NA21234 from project Non-Analyst Project in family 14 inserted into matchbox on Feb 05, 2019, with seqr link /project/R0004_non_analyst_project/family_page/F000014_14/matchmaker_exchange'
+ match3 = 'seqr ID NA21234 from project Non-Analyst Project in family fam14 inserted into matchbox on Feb 05, 2019, with seqr link /project/R0004_non_analyst_project/family_page/F000014_14/matchmaker_exchange'
mock_post_to_slack.assert_called_with('matchmaker_matches', message_template.format(
matches='\n'.join([match1, match2, match3]),
diff --git a/seqr/fixtures/1kg_project.json b/seqr/fixtures/1kg_project.json
index a6e9ec55f3..4a9e17b50b 100644
--- a/seqr/fixtures/1kg_project.json
+++ b/seqr/fixtures/1kg_project.json
@@ -375,7 +375,7 @@
"created_by": null,
"last_modified_date": "2017-03-12T22:37:17.555Z",
"project": 4,
- "family_id": "14",
+ "family_id": "fam14",
"analysis_status": "Rncc",
"success_story": "Differential treatement",
"success_story_types": ["A", "D"]
diff --git a/seqr/management/commands/deactivate_project_search.py b/seqr/management/commands/deactivate_project_search.py
index ea758a3ed2..57088c089e 100644
--- a/seqr/management/commands/deactivate_project_search.py
+++ b/seqr/management/commands/deactivate_project_search.py
@@ -1,6 +1,8 @@
from django.core.management.base import BaseCommand, CommandError
+from clickhouse_search.search import delete_clickhouse_project
from seqr.models import Project, Sample
+from seqr.utils.search.utils import backend_specific_call
from seqr.views.utils.airflow_utils import is_airflow_enabled, trigger_airflow_dag, DELETE_PROJECTS_DAG_NAME
import logging
@@ -21,8 +23,22 @@ def handle(self, *args, **options):
logger.info(f'Deactivated {len(updated)} samples')
- if updated and is_airflow_enabled():
- dataset_types = Sample.objects.filter(guid__in=updated).values_list('dataset_type', flat=True).distinct()
- for dataset_type in dataset_types:
- trigger_airflow_dag(DELETE_PROJECTS_DAG_NAME, project, 'SNV_INDEL')
- logger.info(f'Successfully triggered {DELETE_PROJECTS_DAG_NAME} DAG for {dataset_type} {project.guid}')
+ if updated:
+ dataset_types = Sample.objects.filter(guid__in=updated).values_list('dataset_type', 'sample_type').order_by('dataset_type').distinct()
+ for dataset_type, sample_type in dataset_types:
+ backend_specific_call(
+ lambda *args, **kwargs: True, self._trigger_delete_dag, self._delete_clickhouse_project,
+ )(project, dataset_type, sample_type)
+
+ @staticmethod
+ def _trigger_delete_dag(project, dataset_type, sample_type):
+ if is_airflow_enabled():
+ trigger_airflow_dag(DELETE_PROJECTS_DAG_NAME, project, 'SNV_INDEL')
+ logger.info(f'Successfully triggered {DELETE_PROJECTS_DAG_NAME} DAG for {dataset_type} {project.guid}')
+
+ @staticmethod
+ def _delete_clickhouse_project(project, dataset_type, sample_type):
+ if dataset_type == Sample.DATASET_TYPE_SV_CALLS and sample_type == Sample.SAMPLE_TYPE_WES:
+ dataset_type = 'GCNV'
+ info = delete_clickhouse_project(project, dataset_type)
+ logger.info(info)
diff --git a/seqr/management/commands/reload_clinvar_all_variants.py b/seqr/management/commands/reload_clinvar_all_variants.py
index ab9ddce7a7..3b05a63883 100644
--- a/seqr/management/commands/reload_clinvar_all_variants.py
+++ b/seqr/management/commands/reload_clinvar_all_variants.py
@@ -260,10 +260,13 @@ def handle(self, *args, **options):
return
logger.info(f'Updating Clinvar ClickHouse tables to {new_version} from {existing_version_obj and existing_version_obj.version}.')
# Drop any currently existing variants in the table that may exist due to a
- # previously failed partial run.
- clinvar_run_sql(
- Template(f"ALTER TABLE `$reference_genome/$dataset_type/clinvar_all_variants` DROP PARTITION '{new_version}';")
- )
+ # previously failed partial run. Note that we validate that the Postgresql existing version
+ # is present in ClickHouse to account for the situation where Postgresql has an incorrect
+ # version.
+ if existing_version_obj and ClinvarAllVariantsSnvIndel.objects.filter(version=existing_version_obj.version).exists():
+ clinvar_run_sql(
+ Template(f"ALTER TABLE `$reference_genome/$dataset_type/clinvar_all_variants` DROP PARTITION '{new_version}';")
+ )
# Handle parsing variants
if event == 'end' and elem.tag == 'VariationArchive' and new_version:
diff --git a/seqr/management/commands/reload_saved_variant_json.py b/seqr/management/commands/reload_saved_variant_json.py
index 5b38509d02..1f76eb2fa4 100644
--- a/seqr/management/commands/reload_saved_variant_json.py
+++ b/seqr/management/commands/reload_saved_variant_json.py
@@ -32,6 +32,6 @@ def handle(self, *args, **options):
logging.info("Processing all %s projects" % len(projects))
family_ids = [family_guid] if family_guid else None
- project_list = [(*project, family_ids) for project in projects.values_list('id', 'guid', 'name', 'genome_version')]
+ project_list = [(*project, family_ids) for project in projects.order_by('id').values_list('id', 'guid', 'name', 'genome_version')]
update_projects_saved_variant_json(project_list, user_email='manage_command')
logger.info("Done")
diff --git a/seqr/management/commands/transfer_families_to_different_project.py b/seqr/management/commands/transfer_families_to_different_project.py
index 95ea20879d..27a44c6a85 100644
--- a/seqr/management/commands/transfer_families_to_different_project.py
+++ b/seqr/management/commands/transfer_families_to_different_project.py
@@ -12,6 +12,7 @@
def _disable_search(families, from_project):
search_samples = Sample.objects.filter(is_active=True, individual__family__in=families)
+ updated_family_dataset_types = None
if search_samples:
updated_families = search_samples.values_list("individual__family__family_id", flat=True).distinct()
updated_family_dataset_types = list(search_samples.values_list('dataset_type', 'individual__family__guid').distinct())
@@ -20,9 +21,10 @@ def _disable_search(families, from_project):
logger.info(
f'Disabled search for {num_updated} samples in the following {len(updated_families)} families: {family_summary}'
)
- _trigger_delete_families_dags(from_project, updated_family_dataset_types)
+ return updated_family_dataset_types
-def _trigger_delete_families_dags(from_project, updated_family_dataset_types):
+def _disable_search_trigger_delete_families_dags(families, from_project):
+ updated_family_dataset_types = _disable_search(families, from_project)
updated_families_by_dataset_type = defaultdict(list)
for dataset_type, family_guid in updated_family_dataset_types:
updated_families_by_dataset_type[dataset_type].append(family_guid)
@@ -52,7 +54,17 @@ def handle(self, *args, **options):
missing_id_message = '' if num_found == num_expected else f' No match for: {", ".join(set(family_ids) - set([f.family_id for f in families]))}.'
logger.info(f'Found {num_found} out of {num_expected} families.{missing_id_message}')
- backend_specific_call(lambda *args: None, _disable_search, _disable_search)(families, from_project)
+ found_families = families
+ families = families.filter(analysisgroup__isnull=True)
+ if len(families) < num_found:
+ update_family_ids = set([f.family_id for f in families])
+ group_families = [
+ f'{f.family_id} ({", ".join(f.analysisgroup_set.values_list("name", flat=True))})'
+ for f in found_families if f.family_id not in update_family_ids
+ ]
+ logger.info(f'Skipping {num_found - len(families)} families with analysis groups in the project: {", ".join(group_families)}')
+
+ backend_specific_call(lambda *args: None, _disable_search_trigger_delete_families_dags, _disable_search)(families, from_project)
for variant_tag_type in VariantTagType.objects.filter(project=from_project):
variant_tags = VariantTag.objects.filter(saved_variants__family__in=families, variant_tag_type=variant_tag_type)
diff --git a/seqr/management/tests/check_bam_cram_paths_tests.py b/seqr/management/tests/check_bam_cram_paths_tests.py
index d8105f1931..89ee7b0aa5 100644
--- a/seqr/management/tests/check_bam_cram_paths_tests.py
+++ b/seqr/management/tests/check_bam_cram_paths_tests.py
@@ -46,8 +46,8 @@ def _check_results(self, did_delete, mock_logger, mock_safe_post_to_slack, mock_
self.assertListEqual(sorted(igv_file_paths), expected_remaining_files)
mock_subprocess.assert_has_calls([
- mock.call('gsutil ls gs://readviz/NA20870.cram', stdout=-1, stderr=-2, shell=True),
- mock.call('gsutil ls gs://datasets-gcnv/NA20870.bed.gz', stdout=-1, stderr=-2, shell=True),
+ mock.call('gsutil ls gs://readviz/NA20870.cram', stdout=-1, stderr=-2, shell=True), # nosec
+ mock.call('gsutil ls gs://datasets-gcnv/NA20870.bed.gz', stdout=-1, stderr=-2, shell=True), # nosec
], any_order=True)
calls = [
diff --git a/seqr/management/tests/check_for_new_samples_from_pipeline_tests.py b/seqr/management/tests/check_for_new_samples_from_pipeline_tests.py
index 53294bd9c5..d46c613e38 100644
--- a/seqr/management/tests/check_for_new_samples_from_pipeline_tests.py
+++ b/seqr/management/tests/check_for_new_samples_from_pipeline_tests.py
@@ -45,7 +45,7 @@
f'We are following up on the request to load data from AnVIL on March 12, 2017.
' \
f'We have loaded 1 new WES samples from the AnVIL workspace {anvil_link} to the corresponding seqr project {seqr_link}.' \
f'
Let us know if you have any questions.
All the best,
The seqr team'
-ANVIL_ERROR_TEXT_EMAIL = f"""Dear seqr user,
+ANVIL_ERROR_TEXT_EMAIL = """Dear seqr user,
We are following up on the request to load data from AnVIL workspace ext-data/empty on March 12, 2017. This request could not be loaded due to the following error(s):
- Missing the following expected contigs:chr17
@@ -679,7 +679,7 @@ def test_command(self, mock_email, mock_temp_dir, mock_max_reload_variants):
"""Encountered the following errors loading Non-Analyst Project:
The following 1 families failed sex check:
-- 14: Sample NA21987 has pedigree sex M but imputed sex F""",
+- fam14: Sample NA21987 has pedigree sex M but imputed sex F""",
),
mock.call(
'seqr-data-loading',
@@ -1018,7 +1018,7 @@ def _set_empty_loading_files(self):
def _assert_has_expected_empty_list_file_calls(self):
self.mock_subprocess.assert_called_with(
- 'gsutil ls gs://seqr-hail-search-data/v3.1/GRCh37/MITO/runs/*/*', stdout=-1, stderr=-1, shell=True
+ 'gsutil ls gs://seqr-hail-search-data/v3.1/GRCh37/MITO/runs/*/*', stdout=-1, stderr=-1, shell=True # nosec
)
def _set_reloading_loading_files(self):
@@ -1047,7 +1047,7 @@ def _assert_expected_loading_file_calls(self, single_call):
('gsutil mv /mock/tmp/* gs://seqr-hail-search-data/v3.1/GRCh38/SNV_INDEL/runs/manual__2025-01-24/', -2),
]
self.mock_subprocess.assert_has_calls(
- [mock.call(command, stdout=-1, stderr=stderr, shell=True) for (command, stderr) in calls]
+ [mock.call(command, stdout=-1, stderr=stderr, shell=True) for (command, stderr) in calls] # nosec
)
def _additional_loading_logs(self, data_type, version):
diff --git a/seqr/management/tests/deactivate_project_search_tests.py b/seqr/management/tests/deactivate_project_search_tests.py
index 2b97314b92..3a32a408e0 100644
--- a/seqr/management/tests/deactivate_project_search_tests.py
+++ b/seqr/management/tests/deactivate_project_search_tests.py
@@ -5,22 +5,19 @@
from django.core.exceptions import ObjectDoesNotExist
from django.core.management import call_command
from django.core.management.base import CommandError
-from seqr.models import Sample
-from seqr.views.utils.test_utils import AirflowTestCase
+
+from clickhouse_search.models import EntriesSnvIndel, EntriesMito, EntriesGcnv, ProjectGtStatsSnvIndel, \
+ ProjectGtStatsMito, AnnotationsSnvIndel, AnnotationsMito
+from seqr.models import Sample, Project
+from seqr.views.utils.test_utils import AirflowTestCase, AnvilAuthenticationTestCase, AuthenticationTestCase
PROJECT_GUID = 'R0001_1kg'
VARIANT_ID = '21-3343353-GAGA-G'
-class DeactivateProjectSearchTest(AirflowTestCase):
- fixtures = ['users', '1kg_project']
+class DeactivateProjectSearchTest(object):
- DAG_NAME = 'DELETE_PROJECTS'
- DAG_VARIABLES = {
- 'projects_to_run': [PROJECT_GUID],
- 'dataset_type': 'SNV_INDEL',
- 'reference_genome': 'GRCh37',
- }
+ DELETE_SUCCESS_LOGS = []
@responses.activate
@mock.patch('seqr.management.commands.deactivate_project_search.input')
@@ -37,6 +34,7 @@ def test_command(self, mock_input):
self.assertEqual(str(e.exception), 'Error: user did not confirm')
# Test success
+ Project.objects.filter(guid=PROJECT_GUID).update(genome_version='38')
mock_input.return_value = 'y'
self.reset_logs()
self.maxDiff = None
@@ -49,23 +47,84 @@ def test_command(self, mock_input):
'updateType': 'bulk_update'},
}),
('Deactivated 11 samples', None),
- ('Successfully triggered DELETE_PROJECTS DAG for MITO R0001_1kg', None),
- ('Successfully triggered DELETE_PROJECTS DAG for SV R0001_1kg', None),
- ('Successfully triggered DELETE_PROJECTS DAG for SNV_INDEL R0001_1kg', None),
- ])
+ ] + self.DELETE_SUCCESS_LOGS)
active_samples = Sample.objects.filter(individual__family__project__guid=PROJECT_GUID, is_active=True)
self.assertEqual(active_samples.count(), 0)
- self.assert_airflow_calls(self.DAG_VARIABLES, 5)
+ self._assert_expected_delete()
# Re-running has no effect
self.reset_logs()
call_command('deactivate_project_search', PROJECT_GUID)
self.assert_json_logs(user=None, expected=[('Deactivated 0 samples', None)])
+ def _assert_expected_delete(self):
+ pass
+
+class ElasticsearchDeactivateProjectSearchTest(AuthenticationTestCase, DeactivateProjectSearchTest):
+ fixtures = ['users', '1kg_project']
+
+
+class ClickhouseDeactivateProjectSearchTest(AnvilAuthenticationTestCase, DeactivateProjectSearchTest):
+ fixtures = ['users', '1kg_project', 'reference_data', 'clickhouse_search']
+
+ DELETE_SUCCESS_LOGS = [
+ ('Deleted all MITO search data for project 1kg project nåme with uniçøde', None),
+ ('Deleted all SNV_INDEL search data for project 1kg project nåme with uniçøde', None),
+ ('Deleted all GCNV search data for project 1kg project nåme with uniçøde', None),
+ ]
+
+ def _assert_expected_delete(self):
+ self.assertEqual(EntriesSnvIndel.objects.filter(project_guid=PROJECT_GUID).count(), 0)
+ self.assertEqual(ProjectGtStatsSnvIndel.objects.filter(project_guid=PROJECT_GUID).count(), 0)
+ self.assertEqual(EntriesMito.objects.filter(project_guid=PROJECT_GUID).count(), 0)
+ self.assertEqual(ProjectGtStatsMito.objects.filter(project_guid=PROJECT_GUID).count(), 0)
+ self.assertEqual(EntriesGcnv.objects.filter(project_guid=PROJECT_GUID).count(), 0)
+
+ updated_seqr_pops_by_key = dict(AnnotationsSnvIndel.objects.all().join_seqr_pop().values_list('key', 'seqrPop'))
+ self.assertDictEqual(updated_seqr_pops_by_key, {
+ 1: (2, 2, 1, 1),
+ 2: (1, 1, 0, 0),
+ 3: (0, 0, 0, 0),
+ 4: (0, 0, 0, 0),
+ 5: (1, 1, 0, 0),
+ 6: (0, 0, 0, 0),
+ 22: (0, 3, 0, 1),
+ })
+
+ updated_mito_pops_by_key = dict(AnnotationsMito.objects.all().join_seqr_pop().values_list('key', 'seqrPop'))
+ self.assertDictEqual(updated_mito_pops_by_key, {
+ 6: (0, 0, 0, 0),
+ 7: (0, 0, 0, 0),
+ 8: (0, 0, 0, 0),
+ })
+
+
+class HailBackendDeactivateProjectSearchTest(AirflowTestCase, DeactivateProjectSearchTest):
+ fixtures = ['users', '1kg_project']
+
+ CLICKHOUSE_HOSTNAME = ''
+ DAG_NAME = 'DELETE_PROJECTS'
+ DAG_VARIABLES = {
+ 'projects_to_run': [PROJECT_GUID],
+ 'dataset_type': 'SNV_INDEL',
+ 'reference_genome': 'GRCh38',
+ }
+
+ DELETE_SUCCESS_LOGS = [
+ ('Successfully triggered DELETE_PROJECTS DAG for MITO R0001_1kg', None),
+ ('Successfully triggered DELETE_PROJECTS DAG for SNV_INDEL R0001_1kg', None),
+ ('Successfully triggered DELETE_PROJECTS DAG for SV R0001_1kg', None),
+ ]
+
+ def _assert_expected_delete(self):
+ self.assert_airflow_calls(self.DAG_VARIABLES, 5)
+
def _add_update_check_dag_responses(self, **kwargs):
return self._add_check_dag_variable_responses(self.DAG_VARIABLES, **kwargs)
def _assert_update_check_airflow_calls(self, call_count, offset, update_check_path):
variables_update_check_path = f'{self.MOCK_AIRFLOW_URL}/api/v1/variables/{self.DAG_NAME}'
super()._assert_update_check_airflow_calls(call_count, offset, variables_update_check_path)
+
+
diff --git a/seqr/management/tests/reload_clinvar_all_variants_tests.py b/seqr/management/tests/reload_clinvar_all_variants_tests.py
index 9ffa96ae02..c1056d7d44 100644
--- a/seqr/management/tests/reload_clinvar_all_variants_tests.py
+++ b/seqr/management/tests/reload_clinvar_all_variants_tests.py
@@ -14,7 +14,6 @@
)
from reference_data.models import DataVersions
from seqr.management.commands.reload_clinvar_all_variants import BATCH_SIZE, WEEKLY_XML_RELEASE
-from seqr.views.utils.test_utils import DifferentDbTransactionSupportMixin
WEEKLY_XML_RELEASE_HEADER = ''''''
WEEKLY_XML_RELEASE_DATA = WEEKLY_XML_RELEASE_HEADER + '''
@@ -83,7 +82,7 @@
@mock.patch('seqr.management.commands.reload_clinvar_all_variants.safe_post_to_slack')
@mock.patch('seqr.management.commands.reload_clinvar_all_variants.logger.info')
-class ReloadClinvarAllVariantsTest(DifferentDbTransactionSupportMixin, TestCase):
+class ReloadClinvarAllVariantsTest(TestCase):
databases = '__all__'
fixtures = ['clinvar_all_variants']
diff --git a/seqr/management/tests/transfer_families_to_different_project_tests.py b/seqr/management/tests/transfer_families_to_different_project_tests.py
index de7d9e7db0..b4a93735d4 100644
--- a/seqr/management/tests/transfer_families_to_different_project_tests.py
+++ b/seqr/management/tests/transfer_families_to_different_project_tests.py
@@ -1,21 +1,25 @@
import responses
from django.core.management import call_command
-import mock
-import json
from seqr.models import Family, VariantTagType, VariantTag, Sample
-from seqr.views.utils.test_utils import AirflowTestCase, AuthenticationTestCase
+from seqr.views.utils.test_utils import AirflowTestCase, AuthenticationTestCase, AnvilAuthenticationTestCase
class TransferFamiliesTest(object):
- def _test_command(self, additional_family, logs):
+ DEACTIVATE_SEARCH = True
+ LOGS = []
+
+ def test_command(self):
call_command(
- 'transfer_families_to_different_project', '--from-project=R0001_1kg', '--to-project=R0003_test', additional_family, '2',
+ 'transfer_families_to_different_project', '--from-project=R0001_1kg', '--to-project=R0003_test', '2', '4', '5', '12',
)
+ self.maxDiff = None
self.assert_json_logs(user=None, expected=[
- *logs,
+ ('Found 3 out of 4 families. No match for: 12.', None),
+ ('Skipping 1 families with analysis groups in the project: 5 (Test Group 1)', None),
+ *self.LOGS,
('Updating "Excluded" tags', None),
('Updating families', None),
('Done.', None),
@@ -35,23 +39,46 @@ def _test_command(self, additional_family, logs):
self.assertEqual(len(new_tags), 1)
self.assertEqual(new_tags[0].saved_variants.first().family, family)
- return family
+ samples = Sample.objects.filter(individual__family=family)
+ self.assertEqual(samples.count(), 7)
+ self.assertEqual(samples.filter(is_active=True).count(), 0 if self.DEACTIVATE_SEARCH else 7)
+
+ family = Family.objects.get(family_id='4')
+ self.assertEqual(family.project.guid, 'R0003_test')
+ self.assertEqual(family.individual_set.count(), 1)
class TransferFamiliesLocalTest(TransferFamiliesTest, AuthenticationTestCase):
fixtures = ['users', '1kg_project']
+ DEACTIVATE_SEARCH = False
- def test_es_command(self):
- self._test_command(
- additional_family='12', logs=[('Found 1 out of 2 families. No match for: 12.', None)]
- )
+
+class TransferFamiliesClickhouseTest(TransferFamiliesTest, AnvilAuthenticationTestCase):
+ fixtures = ['users', '1kg_project']
+
+ ES_HOSTNAME = ''
+ LOGS = [
+ ('Disabled search for 7 samples in the following 1 families: 2', None),
+ ]
class TransferFamiliesAirflowTest(TransferFamiliesTest, AirflowTestCase):
fixtures = ['users', '1kg_project']
PROJECT_GUID = 'R0001_1kg' # from-project
DAG_NAME = 'DELETE_FAMILIES'
+ ES_HOSTNAME = ''
+ CLICKHOUSE_HOSTNAME = ''
+
+ LOGS = [
+ ('Disabled search for 7 samples in the following 1 families: 2', None),
+ ('Successfully triggered DELETE_FAMILIES DAG for 1 MITO families', None),
+ ('Successfully triggered DELETE_FAMILIES DAG for 1 SNV_INDEL families', None),
+ ('400 Client Error: Bad Request for url: http://testairflowserver/api/v1/variables/DELETE_FAMILIES', {
+ 'severity': 'ERROR',
+ '@type': 'type.googleapis.com/google.devtools.clouderrorreporting.v1beta1.ReportedErrorEvent',
+ })
+ ]
def setUp(self):
super().setUp()
@@ -88,25 +115,6 @@ def _assert_update_check_airflow_calls(self, call_count, offset, update_check_pa
super()._assert_update_check_airflow_calls(call_count, offset, variables_update_check_path)
@responses.activate
- @mock.patch('seqr.utils.search.elasticsearch.es_utils.ELASTICSEARCH_SERVICE_HOSTNAME', '')
- def test_hail_backend_command(self):
- searchable_family = self._test_command(additional_family='4', logs=[
- ('Found 2 out of 2 families.', None),
- ('Disabled search for 7 samples in the following 1 families: 2', None),
- ('Successfully triggered DELETE_FAMILIES DAG for 1 MITO families', None),
- ('Successfully triggered DELETE_FAMILIES DAG for 1 SNV_INDEL families', None),
- ('400 Client Error: Bad Request for url: http://testairflowserver/api/v1/variables/DELETE_FAMILIES', {
- 'severity': 'ERROR',
- '@type': 'type.googleapis.com/google.devtools.clouderrorreporting.v1beta1.ReportedErrorEvent',
- })
- ])
-
- samples = Sample.objects.filter(individual__family=searchable_family)
- self.assertEqual(samples.count(), 7)
- self.assertEqual(samples.filter(is_active=True).count(), 0)
-
- family = Family.objects.get(family_id='4')
- self.assertEqual(family.project.guid, 'R0003_test')
- self.assertEqual(family.individual_set.count(), 1)
-
+ def test_command(self):
+ super().test_command()
self.assert_airflow_delete_families_calls()
diff --git a/seqr/models.py b/seqr/models.py
index 180ad8e04f..4f4951e90d 100644
--- a/seqr/models.py
+++ b/seqr/models.py
@@ -6,7 +6,7 @@
from django.contrib.postgres.fields import ArrayField
from django.core.exceptions import PermissionDenied, ValidationError
from django.db import models
-from django.db.models import base, options, ForeignKey, JSONField, prefetch_related_objects
+from django.db.models import base, options, prefetch_related_objects
from django.utils import timezone
from django.utils.text import slugify as __slugify
@@ -340,7 +340,7 @@ class Family(ModelWithGUID):
description = models.TextField(null=True, blank=True)
pedigree_image = models.ImageField(null=True, blank=True, upload_to='pedigree_images')
- pedigree_dataset = JSONField(null=True, blank=True)
+ pedigree_dataset = models.JSONField(null=True, blank=True)
assigned_analyst = models.ForeignKey(User, null=True, on_delete=models.SET_NULL,
related_name='assigned_families') # type: ForeignKey
@@ -639,19 +639,19 @@ class Individual(ModelWithGUID):
expected_inheritance = ArrayField(models.CharField(max_length=1, choices=INHERITANCE_CHOICES), null=True)
# features are objects with an id field for HPO id and optional notes and qualifiers fields
- features = JSONField(null=True)
- absent_features = JSONField(null=True)
+ features = models.JSONField(null=True)
+ absent_features = models.JSONField(null=True)
# nonstandard_features are objects with an id field for a free text label and optional
# notes, qualifiers, and categories fields
- nonstandard_features = JSONField(null=True)
- absent_nonstandard_features = JSONField(null=True)
+ nonstandard_features = models.JSONField(null=True)
+ absent_nonstandard_features = models.JSONField(null=True)
# Disorders are a list of MIM IDs
disorders = ArrayField(models.CharField(max_length=10), null=True)
# genes are objects with required key gene (may be blank) and optional key comments
- candidate_genes = JSONField(null=True)
- rejected_genes = JSONField(null=True)
+ candidate_genes = models.JSONField(null=True)
+ rejected_genes = models.JSONField(null=True)
ar_fertility_meds = models.BooleanField(null=True)
ar_iui = models.BooleanField(null=True)
@@ -661,10 +661,10 @@ class Individual(ModelWithGUID):
ar_donoregg = models.BooleanField(null=True)
ar_donorsperm = models.BooleanField(null=True)
- filter_flags = JSONField(null=True)
- pop_platform_filters = JSONField(null=True)
+ filter_flags = models.JSONField(null=True)
+ pop_platform_filters = models.JSONField(null=True)
population = models.CharField(max_length=5, null=True)
- sv_flags = JSONField(null=True)
+ sv_flags = models.JSONField(null=True)
def __unicode__(self):
return self.individual_id.strip()
@@ -839,11 +839,11 @@ class SavedVariant(ModelWithGUID):
key = models.PositiveBigIntegerField(null=True, blank=True)
selected_main_transcript_id = models.CharField(max_length=20, null=True)
- saved_variant_json = JSONField(default=dict)
- genotypes = JSONField(default=dict)
+ saved_variant_json = models.JSONField(default=dict)
+ genotypes = models.JSONField(default=dict)
dataset_type = models.CharField(max_length=13, choices=DATASET_TYPE_CHOICES, null=True, blank=True)
- acmg_classification = JSONField(null=True) # ACMG based classification
+ acmg_classification = models.JSONField(null=True) # ACMG based classification
def __unicode__(self):
chrom, pos = get_chrom_pos(self.xpos)
@@ -1120,7 +1120,7 @@ class Meta:
class DynamicAnalysisGroup(ModelWithGUID):
project = models.ForeignKey('Project', on_delete=models.CASCADE, null=True, blank=True)
name = models.TextField()
- criteria = JSONField()
+ criteria = models.JSONField()
def __unicode__(self):
return self.name.strip()
@@ -1136,7 +1136,7 @@ class Meta:
class VariantSearch(ModelWithGUID):
name = models.CharField(max_length=200, null=True)
order = models.FloatField(null=True, blank=True)
- search = JSONField()
+ search = models.JSONField()
def __unicode__(self):
return self.name or str(self.id)
diff --git a/seqr/utils/gene_utils.py b/seqr/utils/gene_utils.py
index 06b2572981..1b2b615715 100644
--- a/seqr/utils/gene_utils.py
+++ b/seqr/utils/gene_utils.py
@@ -16,8 +16,8 @@ def get_gene(gene_id, user):
return gene_json
-def get_genes(gene_ids, genome_version=None):
- return _get_genes(gene_ids, genome_version=genome_version)
+def get_genes(gene_ids, genome_version=None, **kwargs):
+ return _get_genes(gene_ids, genome_version=genome_version, **kwargs)
def get_genes_for_variant_display(gene_ids, genome_version):
@@ -32,13 +32,13 @@ def get_genes_with_detail(gene_ids, user):
return _get_genes(gene_ids, user=user, gene_fields=ALL_GENE_FIELDS)
-def _get_genes(gene_ids, user=None, gene_fields=None, genome_version=None):
+def _get_genes(gene_ids, user=None, genome_version=None, **kwargs):
gene_filter = {}
_add_genome_version_filter(gene_filter, genome_version)
if gene_ids is not None:
gene_filter['gene_id__in'] = gene_ids
genes = GeneInfo.objects.filter(**gene_filter)
- return {gene['geneId']: gene for gene in _get_json_for_genes(genes, user=user, gene_fields=gene_fields)}
+ return {gene['geneId']: gene for gene in _get_json_for_genes(genes, user=user, **kwargs)}
def _add_genome_version_filter(gene_filter, genome_version):
@@ -124,7 +124,7 @@ def _add_mgi(gene):
}
ALL_GENE_FIELDS.update(VARIANT_GENE_FIELDS)
-def _get_json_for_genes(genes, user=None, gene_fields=None):
+def _get_json_for_genes(genes, user=None, gene_fields=None, **kwargs):
if not gene_fields:
gene_fields = {}
@@ -155,10 +155,10 @@ def _process_result(result, gene):
'{}_set'.format(model.__name__.lower()),
queryset=model.objects.only('gene__gene_id', *model._meta.json_fields)))
- return _get_json_for_models(genes, process_result=_process_result)
+ return _get_json_for_models(genes, process_result=_process_result, **kwargs)
-def parse_locus_list_items(request_json, genome_version=None):
+def parse_locus_list_items(request_json, genome_version=None, **kwargs):
raw_items = request_json.get('rawItems')
if not raw_items:
return None, None, None
@@ -196,6 +196,6 @@ def parse_locus_list_items(request_json, genome_version=None):
gene_symbols_to_ids = get_gene_ids_for_gene_symbols(gene_symbols, genome_version=genome_version)
invalid_items += [symbol for symbol in gene_symbols if not gene_symbols_to_ids.get(symbol)]
gene_ids.update({gene_ids[0] for gene_ids in gene_symbols_to_ids.values() if len(gene_ids)})
- genes_by_id = get_genes(list(gene_ids), genome_version=genome_version) if gene_ids else {}
+ genes_by_id = get_genes(list(gene_ids), genome_version=genome_version, **kwargs) if gene_ids else {}
invalid_items += [gene_id for gene_id in gene_ids if not genes_by_id.get(gene_id)]
return genes_by_id, intervals, invalid_items
\ No newline at end of file
diff --git a/seqr/utils/search/search_utils_tests.py b/seqr/utils/search/search_utils_tests.py
index 681b438f3f..6bfacf5cfc 100644
--- a/seqr/utils/search/search_utils_tests.py
+++ b/seqr/utils/search/search_utils_tests.py
@@ -310,7 +310,7 @@ def test_invalid_search_query_variants(self):
def _test_expected_search_call(self, mock_get_variants, results_cache, search_fields=None, has_gene_search=False,
rs_ids=None, variant_ids=None, parsed_variant_ids=None, inheritance_mode='de_novo',
dataset_type=None, secondary_dataset_type=None, omitted_sample_guids=None,
- exclude_locations=False, exclude=None, annotations=None, annotations_secondary=None, **kwargs):
+ exclude_locations=False, exclude=None, annotations=None, annotations_secondary=None, single_gene_search=False, **kwargs):
expected_search = {
'inheritance_mode': inheritance_mode,
'inheritance_filter': {},
@@ -335,6 +335,9 @@ def _test_expected_search_call(self, mock_get_variants, results_cache, search_fi
if has_gene_search:
gene_ids = ['ENSG00000186092', 'ENSG00000227232']
intervals = [['2', 1234, 5678], ['7', 1, 11100], ['1', 14404, 29570], ['1', 65419, 71585]]
+ if single_gene_search:
+ gene_ids = gene_ids[1:]
+ intervals = intervals[2:3]
self._assert_expected_search_locus(
mock_get_variants.call_args.args[1], dataset_type='MITO_missing' if has_included_gene_search else dataset_type,
gene_ids=gene_ids, intervals=intervals, rs_ids=rs_ids, variant_ids=variant_ids,
@@ -417,15 +420,23 @@ def _mock_get_variants(families, search, user, previous_search_results, genome_v
rs_ids=['rs9876'], variant_ids=[], parsed_variant_ids=[], omitted_sample_guids=SV_SAMPLES, dataset_type='SNV_INDEL',
)
- self.search_model.search['locus']['rawItems'] = 'WASH7P, chr2:1234-5678, chr7:100-10100%10, ENSG00000186092'
+ locus_items = 'WASH7P, chr2:1234-5678, chr7:100-10100%10, ENSG00000186092'
+ self.search_model.search['locus']['rawItems'] = locus_items
query_variants(self.results_model, user=self.user)
self._test_expected_search_call(
mock_get_variants, results_cache, sort='xpos', page=1, num_results=100, skip_genotype_filter=False,
has_gene_search=True,
)
- locus = self.search_model.search.pop('locus')
- self.search_model.search['exclude'] = {'clinvar': ['benign'], 'rawItems': locus['rawItems']}
+ self.search_model.search['locus']['rawItems'] = 'WASH7P'
+ query_variants(self.results_model, user=self.user)
+ self._test_expected_search_call(
+ mock_get_variants, results_cache, sort='xpos', page=1, num_results=100, skip_genotype_filter=False,
+ has_gene_search=True, single_gene_search=True,
+ )
+
+ del self.search_model.search['locus']
+ self.search_model.search['exclude'] = {'clinvar': ['benign'], 'rawItems': locus_items}
query_variants(self.results_model, user=self.user)
self._test_expected_search_call(
mock_get_variants, results_cache, sort='xpos', page=1, num_results=100, skip_genotype_filter=False,
@@ -539,7 +550,7 @@ def _assert_expected_search_locus(self, search_body, dataset_type, gene_ids=None
intervals = [
{'chrom': '2', 'start': 1234, 'end': 5678, 'offset': None},
{'chrom': '7', 'start': 100, 'end': 10100, 'offset': 0.1},
- ]
+ ] if len(gene_ids) > 1 else []
dataset_type = None if dataset_type == 'MITO_missing' else dataset_type
super()._assert_expected_search_locus(
@@ -549,9 +560,10 @@ def _assert_expected_search_locus(self, search_body, dataset_type, gene_ids=None
if gene_ids:
parsed_genes = search_body['parsed_locus']['genes']
for gene in parsed_genes.values():
- self.assertSetEqual(set(gene.keys()), GENE_FIELDS)
+ self.assertSetEqual(set(gene.keys()), {'id', *GENE_FIELDS})
self.assertEqual(parsed_genes['ENSG00000227232']['geneSymbol'], 'WASH7P')
- self.assertEqual(parsed_genes['ENSG00000186092']['geneSymbol'], 'OR4F5')
+ if len(gene_ids) > 1:
+ self.assertEqual(parsed_genes['ENSG00000186092']['geneSymbol'], 'OR4F5')
def _assert_expected_search_samples(self, mock_get_variants, omitted_sample_guids, has_gene_search):
return super()._assert_expected_search_samples(mock_get_variants, omitted_sample_guids, False)
@@ -710,8 +722,11 @@ def _assert_expected_search_locus(self, *args, gene_ids=None, intervals=None, ex
gene_ids = None if exclude_locations else gene_ids
gene_intervals = None
if gene_ids:
- gene_intervals = intervals[2:]
- intervals = intervals[:2]
+ if len(gene_ids) > 1:
+ gene_intervals = {2: intervals[2], 7: intervals[3]}
+ else:
+ gene_intervals = {2: intervals[0]}
+ intervals = intervals[:2] if len(gene_ids) > 1 else None
super()._assert_expected_search_locus(
*args, gene_ids=gene_ids, gene_intervals=gene_intervals, intervals=intervals,
variant_ids=parsed_variant_ids, exclude_intervals=exclude_locations, **kwargs,
diff --git a/seqr/utils/search/utils.py b/seqr/utils/search/utils.py
index 4b1e01d86d..355b098eef 100644
--- a/seqr/utils/search/utils.py
+++ b/seqr/utils/search/utils.py
@@ -324,7 +324,7 @@ def _query_variants(search_model, user, previous_search_results, genome_version,
rs_ids = None
variant_ids = None
parsed_variant_ids = None
- genes, intervals, invalid_items = parse_locus_list_items(locus or exclude, genome_version=genome_version)
+ genes, intervals, invalid_items = parse_locus_list_items(locus or exclude, genome_version=genome_version, additional_model_fields=['id'])
if invalid_items:
raise InvalidSearchException('Invalid genes/intervals: {}'.format(', '.join(invalid_items)))
if not (genes or intervals):
@@ -487,7 +487,7 @@ def _search_dataset_type(search):
lookup_dataset_type = Sample.DATASET_TYPE_VARIANT_CALLS if rsids else _variant_ids_dataset_type(parsed_variant_ids)
return Sample.DATASET_TYPE_VARIANT_CALLS, None, lookup_dataset_type
- intervals = locus['intervals'] if 'exclude_intervals' in locus and not locus['exclude_intervals'] else None
+ intervals = (locus['intervals'] or (locus.get('gene_intervals') or {}).values()) if 'exclude_intervals' in locus and not locus['exclude_intervals'] else None
dataset_type = _annotation_dataset_type(search.get('annotations'), intervals, pathogenicity=search.get('pathogenicity'))
secondary_dataset_type = _annotation_dataset_type(search['annotations_secondary'], intervals) if search.get('annotations_secondary') else None
@@ -623,11 +623,11 @@ def _parse_locus_gene_intervals(genome_version, genes=None, intervals=None, rs_i
intervals = [_format_interval(**interval) for interval in intervals or []]
gene_intervals = gene_ids = None
if genes:
- gene_intervals = sorted([
- [gene[f'{field}Grch{genome_version}'] for field in ['chrom', 'start', 'end']] for gene in genes.values()
- ])
+ gene_intervals = {
+ gene['id']: [gene[f'{field}Grch{genome_version}'] for field in ['chrom', 'start', 'end']] for gene in genes.values()
+ }
if exclude_locations:
- intervals += gene_intervals
+ intervals += sorted(gene_intervals.values())
gene_intervals = None
else:
gene_ids = sorted(genes.keys())
@@ -646,7 +646,7 @@ def _parse_locus_intervals(*args, **kwargs):
parsed_locus = _parse_locus_gene_intervals(*args, **kwargs)
gene_intervals = parsed_locus.pop('gene_intervals')
if gene_intervals:
- parsed_locus['intervals'] = (parsed_locus['intervals'] or []) + gene_intervals
+ parsed_locus['intervals'] = (parsed_locus['intervals'] or []) + sorted(gene_intervals.values())
return parsed_locus
diff --git a/seqr/views/apis/anvil_workspace_api.py b/seqr/views/apis/anvil_workspace_api.py
index 023832e187..9de58abb3f 100644
--- a/seqr/views/apis/anvil_workspace_api.py
+++ b/seqr/views/apis/anvil_workspace_api.py
@@ -3,7 +3,6 @@
import time
from datetime import datetime
from functools import wraps
-from collections import defaultdict
from django.contrib.auth.decorators import user_passes_test
from django.contrib.auth.views import redirect_to_login
@@ -271,9 +270,11 @@ def _trigger_add_workspace_data(project, pedigree_records, user, data_path, samp
# use airflow api to trigger AnVIL dags
reload_summary = f' and {len(previous_loaded_ids)} re-loaded' if previous_loaded_ids else ''
- success_message = f"""
- *{user.email}* requested to load {num_updated_individuals} new{reload_summary} {sample_type} samples ({GENOME_VERSION_LOOKUP.get(project.genome_version)}) from AnVIL workspace *{project.workspace_namespace}/{project.workspace_name}* at
- {data_path} to seqr project <{_get_seqr_project_url(project)}|*{project.name}*> (guid: {project.guid})"""
+ success_message = (
+ f"*{user.email}* requested to load {num_updated_individuals} new{reload_summary} {sample_type} samples "
+ f"({GENOME_VERSION_LOOKUP.get(project.genome_version)}) from AnVIL workspace *{project.workspace_namespace}/{project.workspace_name}* at "
+ f"{data_path} to seqr project <{_get_seqr_project_url(project)}|*{project.name}*> (guid: {project.guid})"
+ )
trigger_success = trigger_airflow_data_loading(
[project], individual_ids, sample_type, Sample.DATASET_TYPE_VARIANT_CALLS, project.genome_version, data_path, user=user, success_message=success_message,
success_slack_channel=SEQR_SLACK_ANVIL_DATA_LOADING_CHANNEL, error_message=f'ERROR triggering AnVIL loading for project {project.guid}',
@@ -291,13 +292,12 @@ def _trigger_add_workspace_data(project, pedigree_records, user, data_path, samp
loading_warning_date = ANVIL_LOADING_DELAY_EMAIL_START_DATE and datetime.strptime(ANVIL_LOADING_DELAY_EMAIL_START_DATE, '%Y-%m-%d')
if loading_warning_date and loading_warning_date <= datetime.now():
try:
- email_body = f"""Hi {user.get_full_name() or user.email},
- We have received your request to load data to seqr from AnVIL. Currently, the Broad Institute is holding an
- internal retreat or closed for the winter break so we may not be able to load data until mid-January
- {loading_warning_date.year + 1}. We appreciate your understanding and support of our research team taking
- some well-deserved time off and hope you also have a nice break.
- - The seqr team
- """
+ email_body = (f"Hi {user.get_full_name() or user.email},\n"
+ "We have received your request to load data to seqr from AnVIL. Currently, the Broad Institute is holding an "
+ "internal retreat or closed for the winter break so we may not be able to load data until mid-January "
+ f"{loading_warning_date.year + 1}. We appreciate your understanding and support of our research team taking "
+ "some well-deserved time off and hope you also have a nice break.\n"
+ "- The seqr team")
send_html_email(email_body, subject='Delay in loading AnVIL in seqr', to=[user.email])
except Exception as e:
logger.error('AnVIL loading delay email error: {}'.format(e), user)
diff --git a/seqr/views/apis/anvil_workspace_api_tests.py b/seqr/views/apis/anvil_workspace_api_tests.py
index 957f687115..ab6c6d60a3 100644
--- a/seqr/views/apis/anvil_workspace_api_tests.py
+++ b/seqr/views/apis/anvil_workspace_api_tests.py
@@ -823,9 +823,7 @@ def _assert_valid_operation(self, project, test_add_data=True):
'sample_source': 'AnVIL',
}
sample_summary = '13 new and 7 re-loaded' if test_add_data else '3 new'
- slack_message = """
- *test_user_manager@test.com* requested to load {sample_summary} WES samples ({version}) from AnVIL workspace *my-seqr-billing/{workspace_name}* at
- gs://test_bucket/test_path.vcf to seqr project (guid: {guid})
+ slack_message = """*test_user_manager@test.com* requested to load {sample_summary} WES samples ({version}) from AnVIL workspace *my-seqr-billing/{workspace_name}* at gs://test_bucket/test_path.vcf to seqr project (guid: {guid})
Pedigree files have been uploaded to gs://seqr-loading-temp/v3.1/{version}/SNV_INDEL/pedigrees/WES
@@ -892,7 +890,7 @@ def _test_mv_file_and_triggering_dag_exception(self, url, workspace, sample_data
slack_message_on_failure = """ERROR triggering AnVIL loading for project {guid}: LOADING_PIPELINE DAG is running and cannot be triggered again.
- DAG LOADING_PIPELINE should be triggered with following:
+ DAG LOADING_PIPELINE should be triggered with following:
```{dag}```
""".format(
guid=project.guid,
@@ -969,12 +967,8 @@ def _test_after_email_date(self, url, request_body):
response = self.client.post(url, content_type='application/json', data=json.dumps(request_body))
self.assertEqual(response.status_code, 200)
self.mock_send_email.assert_called_with("""Hi Test Manager User,
- We have received your request to load data to seqr from AnVIL. Currently, the Broad Institute is holding an
- internal retreat or closed for the winter break so we may not be able to load data until mid-January
- 2022. We appreciate your understanding and support of our research team taking
- some well-deserved time off and hope you also have a nice break.
- - The seqr team
- """, subject='Delay in loading AnVIL in seqr', to=['test_user_manager@test.com'])
+We have received your request to load data to seqr from AnVIL. Currently, the Broad Institute is holding an internal retreat or closed for the winter break so we may not be able to load data until mid-January 2022. We appreciate your understanding and support of our research team taking some well-deserved time off and hope you also have a nice break.
+- The seqr team""", subject='Delay in loading AnVIL in seqr', to=['test_user_manager@test.com'])
self.mock_api_logger.error.assert_called_with(
'AnVIL loading delay email error: Unable to send email', self.manager_user)
diff --git a/seqr/views/apis/data_manager_api_tests.py b/seqr/views/apis/data_manager_api_tests.py
index 8e14871f70..33a6b9bc60 100644
--- a/seqr/views/apis/data_manager_api_tests.py
+++ b/seqr/views/apis/data_manager_api_tests.py
@@ -1495,8 +1495,8 @@ def _has_expected_ped_files(self, mock_open, mock_mkdir, dataset_type, sample_ty
self.assertEqual(len(file), 3)
self.assertListEqual(file, [
pedigree_header,
- ['R0004_non_analyst_project', 'F000014_14', '14', 'NA21234', '', '', 'F'] + (['ABC123'] if has_remap else []),
- ['R0004_non_analyst_project', 'F000014_14', '14', 'NA21987', '', '', 'M'] + ([''] if has_remap else []),
+ ['R0004_non_analyst_project', 'F000014_14', 'fam14', 'NA21234', '', '', 'F'] + (['ABC123'] if has_remap else []),
+ ['R0004_non_analyst_project', 'F000014_14', 'fam14', 'NA21987', '', '', 'M'] + ([''] if has_remap else []),
])
def _test_load_single_project(self, mock_open, mock_mkdir, response, *args, **kwargs):
@@ -1884,7 +1884,7 @@ def _assert_trigger_error(self, response, body, dag_json, **kwargs):
'WGS', 'WES').replace('SNV_INDEL', 'GCNV').replace('v01', 'v3.1')
error_message = f"""ERROR triggering internal WES SV loading: {errors[0]}
- DAG LOADING_PIPELINE should be triggered with following:
+ DAG LOADING_PIPELINE should be triggered with following:
```{dag_json}```
"""
self.mock_slack.assert_called_once_with(SEQR_SLACK_LOADING_NOTIFICATION_CHANNEL, error_message)
@@ -1919,7 +1919,7 @@ def _trigger_error(self, url, body, dag_json, mock_open, mock_mkdir):
'warnings': None,
'errors': [
'The following samples are included in airtable but are missing from the VCF: NA21987',
- 'The following families have previously loaded samples absent from airtable\nFamily 14: NA21234, NA21654',
+ 'The following families have previously loaded samples absent from airtable\nFamily fam14: NA21234, NA21654',
],
})
self.assertEqual(len(responses.calls), 2)
diff --git a/seqr/views/apis/family_api_tests.py b/seqr/views/apis/family_api_tests.py
index eb5450060e..17a1dd9a7e 100644
--- a/seqr/views/apis/family_api_tests.py
+++ b/seqr/views/apis/family_api_tests.py
@@ -459,7 +459,7 @@ def test_update_family_fields(self):
self.assertEqual(response.status_code, 200)
response_json = response.json()
self.assertEqual(response_json['F000014_14']['description'], 'Updated description')
- expected_id = 'new_id' if self._anvil_enabled() else '14'
+ expected_id = 'new_id' if self._anvil_enabled() else 'fam14'
self.assertEqual(response_json['F000014_14'][FAMILY_ID_FIELD], expected_id)
self.assertEqual(response_json['F000014_14']['displayName'], expected_id)
diff --git a/seqr/views/apis/report_api.py b/seqr/views/apis/report_api.py
index b85f2d852a..4d748d14ab 100644
--- a/seqr/views/apis/report_api.py
+++ b/seqr/views/apis/report_api.py
@@ -121,6 +121,7 @@ def anvil_export(request, project_guid):
project = get_project_and_check_permissions(project_guid, request.user)
parsed_rows = defaultdict(list)
+ family_id_map = {}
family_diseases = {}
def _add_row(row, family_id, row_type):
@@ -130,7 +131,7 @@ def _add_row(row, family_id, row_type):
if not (discovery_row.get(GENE_COLUMN) or discovery_row.get('sv_type'))]
if missing_gene_rows:
raise ErrorsWarningsException(
- [f'Discovery variant(s) {", ".join(sorted(missing_gene_rows))} in family {family_id} have no associated gene'])
+ [f'Discovery variant(s) {", ".join(sorted(missing_gene_rows))} in family {family_id_map[family_id]} have no associated gene'])
parsed_rows[row_type] += [{
'entity:discovery_id': f'{discovery_row["chrom"]}_{discovery_row["pos"]}_{discovery_row["participant_id"]}',
**{k: str(discovery_row.get(k.lower()) or '') for k in ['Zygosity', 'Chrom', 'Pos', 'Ref', 'Alt', 'Transcript']},
@@ -164,6 +165,7 @@ def _add_row(row, family_id, row_type):
'disease_id': row.get('condition_id', '').replace('|', ';'),
'disease_description': row.get('known_condition_name', '').replace('|', ';'),
}
+ family_id_map[family_id] = row[id_field]
parsed_rows[row_type].append(row)
max_loaded_date = request.GET.get('loadedBefore') or (datetime.now() - timedelta(days=365)).strftime('%Y-%m-%d')
diff --git a/seqr/views/apis/report_api_tests.py b/seqr/views/apis/report_api_tests.py
index 86350c7a2b..e8c67004d1 100644
--- a/seqr/views/apis/report_api_tests.py
+++ b/seqr/views/apis/report_api_tests.py
@@ -784,7 +784,7 @@ def _check_anvil_export_response(self, response, mock_zip, no_analyst_project_ur
response = self.client.get(no_analyst_project_url)
self.assertEqual(response.status_code, 400)
self.assertEqual(response.json()['errors'],
- ['Discovery variant(s) 1-248367227-TC-T, MT-14783-T-C in family 14 have no associated gene'])
+ ['Discovery variant(s) 1-248367227-TC-T, MT-14783-T-C in family fam14 have no associated gene'])
@mock.patch('seqr.views.apis.report_api.GREGOR_DATA_MODEL_URL', MOCK_DATA_MODEL_URL)
@mock.patch('seqr.views.apis.report_api.datetime')
diff --git a/seqr/views/apis/summary_data_api_tests.py b/seqr/views/apis/summary_data_api_tests.py
index 5b1393ca6a..a917d78829 100644
--- a/seqr/views/apis/summary_data_api_tests.py
+++ b/seqr/views/apis/summary_data_api_tests.py
@@ -129,8 +129,8 @@
EXPECTED_NO_GENE_SAMPLE_METADATA_ROW = {
'participant_id': 'NA21234',
'familyGuid': 'F000014_14',
- 'family_id': '14',
- 'displayName': '14',
+ 'family_id': 'fam14',
+ 'displayName': 'fam14',
'projectGuid': 'R0004_non_analyst_project',
'internal_project_id': 'Non-Analyst Project',
'affected_status': 'Affected',
@@ -406,6 +406,14 @@ def test_saved_variants_page(self):
if 'totalSampleCounts' in response_json:
self.assertDictEqual(response_json['totalSampleCounts'], {'MITO': {'WES': 1}, 'SNV_INDEL': {'WES': 7}, 'SV': {'WES': 3, 'WGS': 3}})
+ all_tag_url = reverse(saved_variants_page, args=['ALL'])
+ response = self.client.get('{}?gene=ENSG00000135953'.format(all_tag_url))
+ self.assertEqual(response.status_code, 200)
+ report_variants = {'SV0027168_191912632_r0384_rare', 'SV0027167_191912633_r0384_rare'}
+ self.assertSetEqual(set(response.json()['savedVariantsByGuid'].keys()), {
+ 'SV0000002_1248367227_r0390_100', 'SV0000013_prefix_19107_DEL_r00', *report_variants, *expected_variant_guids,
+ })
+
# Test analyst behavior
self.login_analyst_user()
response = self.client.get(gene_url)
@@ -414,13 +422,6 @@ def test_saved_variants_page(self):
self.assertSetEqual(set(response_json.keys()), self.SAVED_VARIANT_RESPONSE_KEYS)
self.assertSetEqual(set(response_json['savedVariantsByGuid'].keys()), expected_variant_guids)
- all_tag_url = reverse(saved_variants_page, args=['ALL'])
- response = self.client.get('{}?gene=ENSG00000135953'.format(all_tag_url))
- self.assertEqual(response.status_code, 200)
- expected_variant_guids.add('SV0000002_1248367227_r0390_100')
- report_variants = {'SV0027168_191912632_r0384_rare', 'SV0027167_191912633_r0384_rare'}
- self.assertSetEqual(set(response.json()['savedVariantsByGuid'].keys()), {*report_variants, *expected_variant_guids})
-
multi_tag_url = reverse(saved_variants_page, args=['Review;Tier 1 - Novel gene and phenotype'])
response = self.client.get('{}?gene=ENSG00000135953'.format(multi_tag_url))
self.assertEqual(response.status_code, 200)
@@ -872,7 +873,7 @@ def test_mme_details(self, *args):
def test_saved_variants_page(self):
super(AnvilSummaryDataAPITest, self).test_saved_variants_page()
assert_has_expected_calls(self, [
- self.no_access_user, self.manager_user, self.manager_user, self.analyst_user, self.analyst_user
+ self.no_access_user, self.manager_user, self.manager_user, self.manager_user, self.analyst_user, self.analyst_user
], skip_group_call_idxs=[2])
self.mock_get_ws_access_level.assert_called_with(
self.analyst_user, 'my-seqr-billing', 'anvil-1kg project nåme with uniçøde')
diff --git a/seqr/views/apis/variant_search_api.py b/seqr/views/apis/variant_search_api.py
index ed699a6649..c8d6781bce 100644
--- a/seqr/views/apis/variant_search_api.py
+++ b/seqr/views/apis/variant_search_api.py
@@ -592,31 +592,38 @@ def _update_lookup_variant(variant, response):
no_access_families = set(variant['familyGenotypes']) - set(variant['familyGuids'])
individual_summary_map = {
- (i.pop('family__guid'), i.pop('individual_id')): i
+ (i.pop('family__guid'), i.pop('individual_id')): (i.pop('guid'), i)
for i in Individual.objects.filter(family__guid__in=no_access_families).values(
- 'family__guid', 'individual_id', 'affected', 'sex', 'features',
+ 'family__guid', 'individual_id', 'affected', 'sex', 'features', 'guid',
vlmContactEmail=F('family__project__vlm_contact_email'),
)
}
- add_individual_hpo_details(individual_summary_map.values())
+ add_individual_hpo_details([i for _, i in individual_summary_map.values()])
variant['genotypes'] = {}
variant['lookupFamilyGuids'] = sorted(variant.pop('familyGuids'))
variant['familyGuids'] = []
for family_guid in variant['lookupFamilyGuids']:
- variant['genotypes'].update({
- individual_guid_map[(family_guid, genotype['sampleId'])]: genotype
- for genotype in variant['familyGenotypes'].pop(family_guid)
- })
+ for genotype in variant['familyGenotypes'].pop(family_guid):
+ individual_guid = individual_guid_map[(family_guid, genotype['sampleId'])]
+ if individual_guid in variant['genotypes']:
+ genotype = [variant['genotypes'][individual_guid], genotype]
+ variant['genotypes'][individual_guid] = genotype
for i, (unmapped_family_guid, genotypes) in enumerate(variant.pop('familyGenotypes').items()):
family_guid = f'F{i}_{variant["variantId"]}'
variant['lookupFamilyGuids'].append(family_guid)
if unmapped_family_guid in variant.get('liftedFamilyGuids', []):
variant['liftedFamilyGuids'][variant['liftedFamilyGuids'].index(unmapped_family_guid)] = family_guid
+ individual_guid_map = {}
for j, genotype in enumerate(genotypes):
+ unmapped_individual_guid, individual = individual_summary_map[(genotype.pop('familyGuid'), genotype.pop('sampleId'))]
+ if unmapped_individual_guid in individual_guid_map:
+ individual_guid = individual_guid_map[unmapped_individual_guid]
+ variant['genotypes'][individual_guid] = [variant['genotypes'][individual_guid], genotype]
+ continue
individual_guid = f'I{j}_{family_guid}'
- individual = individual_summary_map[(genotype.pop('familyGuid'), genotype.pop('sampleId'))]
+ individual_guid_map[unmapped_individual_guid] = individual_guid
feature_category_count = defaultdict(int)
for feature in individual['features'] or []:
feature_category_count[feature.get('category', 'Other')] += 1
diff --git a/seqr/views/apis/variant_search_api_tests.py b/seqr/views/apis/variant_search_api_tests.py
index 721d156826..89017e7d2d 100644
--- a/seqr/views/apis/variant_search_api_tests.py
+++ b/seqr/views/apis/variant_search_api_tests.py
@@ -7,7 +7,7 @@
from django.urls.base import reverse
from elasticsearch.exceptions import ConnectionTimeout, TransportError
-from hail_search.test_utils import HAIL_BACKEND_SINGLE_FAMILY_VARIANTS, VARIANT_LOOKUP_VARIANT, GCNV_VARIANT1, SV_VARIANT1
+from clickhouse_search.test_utils import VARIANT2, VARIANT3, VARIANT_LOOKUP_VARIANT, GCNV_VARIANT1, SV_VARIANT1
from seqr.models import VariantSearchResults, LocusList, Project, VariantSearch
from seqr.utils.search.utils import InvalidSearchException
from seqr.utils.search.elasticsearch.es_utils import InvalidIndexException
@@ -25,7 +25,9 @@
SEARCH = {'filters': {}, 'inheritance': None}
PROJECT_FAMILIES = [{'projectGuid': PROJECT_GUID, 'familyGuids': ['F000001_1', 'F000002_2']}]
-VARIANTS_WITH_DISCOVERY_TAGS = deepcopy(VARIANTS + HAIL_BACKEND_SINGLE_FAMILY_VARIANTS)
+ALL_VARIANTS = VARIANTS + [VARIANT2, VARIANT3]
+
+VARIANTS_WITH_DISCOVERY_TAGS = deepcopy(ALL_VARIANTS)
DISCOVERY_TAGS = [{
'savedVariant': {
'variantGuid': 'SV0000006_1248367227_r0003_tes',
@@ -87,7 +89,7 @@
]
EXPECTED_SEARCH_RESPONSE = {
- 'searchedVariants': VARIANTS + HAIL_BACKEND_SINGLE_FAMILY_VARIANTS,
+ 'searchedVariants': ALL_VARIANTS,
'savedVariantsByGuid': {
'SV0000001_2103343353_r0390_100': expected_detail_saved_variant,
'SV0000002_1248367227_r0390_100': EXPECTED_SAVED_VARIANT,
@@ -260,7 +262,7 @@
def _get_es_variants(results_model, **kwargs):
results_model.save()
- return deepcopy(VARIANTS + HAIL_BACKEND_SINGLE_FAMILY_VARIANTS), len(VARIANTS + HAIL_BACKEND_SINGLE_FAMILY_VARIANTS)
+ return deepcopy(ALL_VARIANTS), len(ALL_VARIANTS)
def _get_empty_es_variants(results_model, **kwargs):
@@ -529,15 +531,15 @@ def test_query_variants(self, mock_get_variants, mock_get_gene_counts, mock_erro
['12', '48367227', 'TC', 'T', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '',
'', '2', 'AIP (None)|Known gene for phenotype (None)|Excluded (None)', 'a later note (None)|test n\xf8te (None)', '', '', '', '', '', '',
'', '', '', '', '', '', '', '', '', '', '', ''],
- ['1', '38724419', 'T', 'G', 'ENSG00000177000', 'missense_variant', '28', '0.29499998688697815', '0',
- '0.28899794816970825', '0.24615199863910675', '20.899999618530273', '0.19699999690055847',
- '2.000999927520752', '0.0', '0.1', '0.05', '', '', 'rs1801131', 'ENST00000383791.8:c.156A>C',
+ ['1', '38724419', 'T', 'G', 'ENSG00000177000', 'missense_variant', '10', '0.295', '0',
+ '0.289', '0.24615', '20.9', '0.197',
+ '2.001', '0.0', '0.1', '0.05', '', '', 'rs1801131', 'ENST00000383791.8:c.156A>C',
'ENSP00000373301.3:p.Leu52Phe', 'Conflicting_classifications_of_pathogenicity', '1', '2', '', '', '', '', '', 'HG00731', '2', '', '99', '1.0',
'HG00732', '1', '', '99', '0.625', 'HG00733', '0', '', '40', '0.0'],
- ['1', '91502721', 'G', 'A', 'ENSG00000097046', 'intron_variant', '4', '0.0', '0.38041073083877563', '0.0',
- '0.36268100142478943', '2.753999948501587', '', '1.378000020980835', '0.009999999776482582', '', '', '',
+ ['1', '91502721', 'G', 'A', 'ENSG00000097046', 'intron_variant', '7', '0.0', '0.38041', '0.0',
+ '0.36268', '2.754', '', '1.378', '0.01', '', '', '',
'', 'rs13447464', 'ENST00000234626.11:c.-63-251G>A', '', '', '', '2', '', '', '', '', '', 'HG00731',
- '1', '', '99', '1.0', 'HG00732', '0', '', '99', '0.4594594594594595', 'HG00733', '1', '', '99', '0.4074074074074074'],
+ '1', '', '99', '1.0', 'HG00732', '0', '', '99', '0.45946', 'HG00733', '1', '', '99', '0.40741'],
]
self.assertListEqual([line.split('\t') for line in response.content.decode().strip().split('\n')], expected_content)
@@ -563,16 +565,16 @@ def test_query_variants(self, mock_get_variants, mock_get_gene_counts, mock_erro
['12', '48367227', 'TC', 'T', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '', '',
'2', 'AIP (None)|Known gene for phenotype (None)|Excluded (None)', 'a later note (None)|test n\xf8te (None)',
'', '', '', '', '', '', '', '', '', '', '', '', '', '', '',],
- ['1', '38724419', 'T', 'G', 'ENSG00000177000', 'missense_variant', '28', '0.29499998688697815', '0',
- '0.28899794816970825', '0.24615199863910675', '20.899999618530273', '0.19699999690055847',
- '2.000999927520752', '0.0', '0.1', '0.05', '', '', 'rs1801131', 'ENST00000383791.8:c.156A>C',
+ ['1', '38724419', 'T', 'G', 'ENSG00000177000', 'missense_variant', '10', '0.295', '0',
+ '0.289', '0.24615', '20.9', '0.197',
+ '2.001', '0.0', '0.1', '0.05', '', '', 'rs1801131', 'ENST00000383791.8:c.156A>C',
'ENSP00000373301.3:p.Leu52Phe', 'Conflicting_classifications_of_pathogenicity', '1', '2', '', '', 'HG00731', '2', '', '99', '1.0',
'HG00732', '1', '', '99', '0.625', 'HG00733', '0', '', '40', '0.0'],
- ['1', '91502721', 'G', 'A', 'ENSG00000097046', 'intron_variant', '4', '0.0', '0.38041073083877563', '0.0',
- '0.36268100142478943', '2.753999948501587', '', '1.378000020980835', '0.009999999776482582', '', '',
+ ['1', '91502721', 'G', 'A', 'ENSG00000097046', 'intron_variant', '7', '0.0', '0.38041', '0.0',
+ '0.36268', '2.754', '', '1.378', '0.01', '', '',
'', '', 'rs13447464', 'ENST00000234626.11:c.-63-251G>A', '', '', '', '2', '', '', 'HG00731',
- '1', '', '99', '1.0', 'HG00732', '0', '', '99', '0.4594594594594595', 'HG00733', '1', '', '99',
- '0.4074074074074074'],
+ '1', '', '99', '1.0', 'HG00732', '0', '', '99', '0.45946', 'HG00733', '1', '', '99',
+ '0.40741'],
]
self.assertListEqual([line.split('\t') for line in response.content.decode().strip().split('\n')], expected_content)
@@ -668,7 +670,7 @@ def _get_variants(results_model, **kwargs):
results_model.save()
self.assertSetEqual(expected_searched_families, {f.guid for f in results_model.families.all()})
matched_variants = [
- deepcopy(variant) for variant in VARIANTS + HAIL_BACKEND_SINGLE_FAMILY_VARIANTS
+ deepcopy(variant) for variant in ALL_VARIANTS
if any(family_guid in expected_searched_families for family_guid in variant['familyGuids'])
]
return matched_variants, len(matched_variants)
@@ -914,11 +916,23 @@ def test_variant_lookup(self, mock_variant_lookup):
'lookupFamilyGuids': ['F0_1-10439-AC-A', 'F1_1-10439-AC-A', 'F2_1-10439-AC-A'],
'liftedFamilyGuids': ['F2_1-10439-AC-A'],
'genotypes': {
- 'I0_F0_1-10439-AC-A': {'ab': 0.0, 'dp': 60, 'gq': 20, 'numAlt': 0, 'filters': [], 'sampleType': 'WES'},
- 'I1_F0_1-10439-AC-A': {'ab': 0.0, 'dp': 24, 'gq': 0, 'numAlt': 0, 'filters': [], 'sampleType': 'WES'},
- 'I2_F0_1-10439-AC-A': {'ab': 0.5, 'dp': 10, 'gq': 99, 'numAlt': 1, 'filters': [], 'sampleType': 'WES'},
- 'I0_F1_1-10439-AC-A': {'ab': 1.0, 'dp': 6, 'gq': 16, 'numAlt': 2, 'filters': [], 'sampleType': 'WES'},
- 'I0_F2_1-10439-AC-A': {'ab': 0.531000018119812, 'dp': 27, 'gq': 87, 'numAlt': 1, 'filters': None, 'sampleType': 'WGS'},
+ 'I0_F0_1-10439-AC-A': [
+ {'ab': 0.0, 'dp': 60, 'gq': 20, 'numAlt': 0, 'filters': [], 'sampleType': 'WES'},
+ {'ab': 0.0, 'dp': 60, 'gq': 20, 'numAlt': 0, 'filters': [], 'sampleType': 'WGS'},
+ ],
+ 'I1_F0_1-10439-AC-A': [
+ {'ab': 0.0, 'dp': 24, 'gq': 0, 'numAlt': 0, 'filters': [], 'sampleType': 'WES'},
+ {'ab': 0.0, 'dp': 24, 'gq': 99, 'numAlt': 1, 'filters': [], 'sampleType': 'WGS'},
+ ],
+ 'I2_F0_1-10439-AC-A': [
+ {'ab': 0.5, 'dp': 10, 'gq': 99, 'numAlt': 1, 'filters': [], 'sampleType': 'WES'},
+ {'ab': 0.5, 'dp': 10, 'gq': 99, 'numAlt': 2, 'filters': [], 'sampleType': 'WGS'},
+ ],
+ 'I0_F1_1-10439-AC-A': [
+ {'ab': 1.0, 'dp': 6, 'gq': 16, 'numAlt': 2, 'filters': [], 'sampleType': 'WES'},
+ {'ab': 1.0, 'dp': 6, 'gq': 16, 'numAlt': 2, 'filters': [], 'sampleType': 'WGS'},
+ ],
+ 'I0_F2_1-10439-AC-A': {'ab': 0.531, 'dp': 27, 'gq': 87, 'numAlt': 1, 'filters': [], 'sampleType': 'WGS'},
},
}
del expected_variant['familyGenotypes']
@@ -991,8 +1005,11 @@ def test_variant_lookup(self, mock_variant_lookup):
'lookupFamilyGuids': ['F000002_2', 'F000011_11', 'F000014_14'],
'liftedFamilyGuids': ['F000014_14'],
'genotypes': {
- individual_guid: {**expected_variant['genotypes'][anon_individual_guid], **genotype}
- for individual_guid, anon_individual_guid, genotype in individual_guid_map
+ individual_guid: [
+ {**sample_genotype, **genotype} for sample_genotype in expected_variant['genotypes'][anon_individual_guid]
+ ] if isinstance(expected_variant['genotypes'][anon_individual_guid], list) else {
+ **expected_variant['genotypes'][anon_individual_guid], **genotype,
+ } for individual_guid, anon_individual_guid, genotype in individual_guid_map
},
'genomeVersion': '37',
'variantId': '1-248367227-TC-T',
diff --git a/seqr/views/utils/airflow_utils.py b/seqr/views/utils/airflow_utils.py
index 4a8b7030cd..634903a192 100644
--- a/seqr/views/utils/airflow_utils.py
+++ b/seqr/views/utils/airflow_utils.py
@@ -79,7 +79,7 @@ def _send_load_data_slack_msg(messages: list[str], channel: str, dag: dict):
def _send_slack_msg_on_failure_trigger(e, dag, error_message):
message_content = f"""{error_message}: {e}
- DAG {LOADING_PIPELINE_DAG_NAME} should be triggered with following:
+ DAG {LOADING_PIPELINE_DAG_NAME} should be triggered with following:
```{json.dumps(dag, indent=4)}```
"""
safe_post_to_slack(SEQR_SLACK_LOADING_NOTIFICATION_CHANNEL, message_content)
diff --git a/seqr/views/utils/export_utils.py b/seqr/views/utils/export_utils.py
index bd72a50b93..71261e921c 100644
--- a/seqr/views/utils/export_utils.py
+++ b/seqr/views/utils/export_utils.py
@@ -1,5 +1,3 @@
-from collections import OrderedDict
-import json
import openpyxl as xl
import os
from tempfile import NamedTemporaryFile, TemporaryDirectory
diff --git a/seqr/views/utils/orm_to_json_utils_tests.py b/seqr/views/utils/orm_to_json_utils_tests.py
index 4a539a787a..1b553d3f30 100644
--- a/seqr/views/utils/orm_to_json_utils_tests.py
+++ b/seqr/views/utils/orm_to_json_utils_tests.py
@@ -3,7 +3,7 @@
from copy import deepcopy
from seqr.models import Project, Sample, IgvSample, SavedVariant, VariantNote, LocusList, VariantSearch
from seqr.views.utils.orm_to_json_utils import get_json_for_user, _get_json_for_project, \
- get_json_for_sample, get_json_for_saved_variants, get_json_for_variant_note, get_json_for_locus_list, \
+ get_json_for_sample, get_json_for_variant_note, get_json_for_locus_list, \
get_json_for_saved_searches, get_json_for_saved_variants_with_tags, get_json_for_current_user
from seqr.views.utils.test_utils import AuthenticationTestCase, AnvilAuthenticationTestCase, \
PROJECT_FIELDS, SAMPLE_FIELDS, SAVED_VARIANT_FIELDS, \
diff --git a/seqr/views/utils/variant_utils.py b/seqr/views/utils/variant_utils.py
index 6f35ba9ae1..5f71832f7a 100644
--- a/seqr/views/utils/variant_utils.py
+++ b/seqr/views/utils/variant_utils.py
@@ -19,7 +19,7 @@
from seqr.utils.gene_utils import get_genes_for_variants
from seqr.utils.middleware import ErrorsWarningsException
from seqr.utils.xpos_utils import get_xpos
-from seqr.views.utils.json_to_orm_utils import update_model_from_json, create_model_from_json
+from seqr.views.utils.json_to_orm_utils import create_model_from_json
from seqr.views.utils.orm_to_json_utils import get_json_for_discovery_tags, get_json_for_locus_lists, \
get_json_for_queryset, get_json_for_rna_seq_outliers, get_json_for_saved_variants_with_tags, \
get_json_for_matchmaker_submissions
diff --git a/ui/pages/Public/components/Faq.jsx b/ui/pages/Public/components/Faq.jsx
index b2d0aec7b0..a9d276571d 100644
--- a/ui/pages/Public/components/Faq.jsx
+++ b/ui/pages/Public/components/Faq.jsx
@@ -68,6 +68,12 @@ const FAQS = [
seqr is not designed for cohort or gene burden analyses. You can search for variants in a candidate gene
across your data but seqr will not provide a quantification of how much variation you should expect to see.
+
+ seqr does not support data aligned to any reference genome other than GRCh37 and GRCh38. While both of these
+ builds are available, GRCh38 is recommended whenever possible as it includes the most up-to-date
+ annotations. Additionally, seqr does not support loading data from different builds into the same project,
+ or searching across multiple projects that have different builds.
+
seqr is not an annotation pipeline for a VCF. While annotations are added when data is loaded in seqr, you
cannot output the annotated VCF from seqr.
@@ -88,6 +94,12 @@ const FAQS = [
candidato a través de sus datos, pero seqr no proporcionará una cuantificación de cuánta variación debe
esperar ver.
+
+ seqr no admite datos alineados con ningún genoma de referencia que no sean GRCh37 y GRCh38. Si bien ambas
+ compilaciones están disponibles, se recomienda GRCh38 siempre que sea posible, ya que incluye las
+ anotaciones más actualizadas. Además, seqr no permite cargar datos de diferentes compilaciones en el mismo
+ proyecto ni realizar búsquedas en varios proyectos con compilaciones diferentes.
+
seqr no es un canal de anotación para un VCF. Aunque las anotaciones se agregan cuando los datos se cargan
en seqr, no se puede generar el VCF anotado de seqr.