Skip to content

Commit 39e10f1

Browse files
hotfix: log full name of directory if not found (#225)
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
1 parent fc4ee6e commit 39e10f1

17 files changed

Lines changed: 153 additions & 344 deletions

main/COMO.ipynb

Lines changed: 27 additions & 27 deletions
Original file line numberDiff line numberDiff line change
@@ -53,7 +53,7 @@
5353
" 3. Compare results between perturbed and unperturbed models (i.e., knocked-out models vs non-knocked-out models)\n",
5454
" 4. Integrate with disease genes and create a score of drug targets"
5555
],
56-
"id": "ea1c9614518c30bc"
56+
"id": "90fba8b56a999517"
5757
},
5858
{
5959
"metadata": {},
@@ -223,7 +223,7 @@
223223
" </tbody>\n",
224224
"</table>"
225225
],
226-
"id": "28d59417491e63db"
226+
"id": "dcc8d38caab63fab"
227227
},
228228
{
229229
"metadata": {},
@@ -254,7 +254,7 @@
254254
"scrna_matrix_filepath = {context: Path(f\"data/results/{context}/scrna/scrna_{context}.csv\") for context in context_names}\n",
255255
"# scrna_matrix_filepath = [Path(f\"data/results/{context}/scrna/scrna_{context}.h5ad\") for context in context_names]\n"
256256
],
257-
"id": "18061578964ca096"
257+
"id": "fd4c387eab48e9ee"
258258
},
259259
{
260260
"metadata": {},
@@ -266,7 +266,7 @@
266266
"- `taxon_id`: The [NCBI Taxon ID](https://www.ncbi.nlm.nih.gov/taxonomy) to use\n",
267267
"- `preprocess_mode`: This should be set to `\"create-matrix\"` if you are **not** providing a matrix, otherwise set it to `\"provide-matrix\"`"
268268
],
269-
"id": "b475e41b000d0cb5"
269+
"id": "24a8475316ff6d70"
270270
},
271271
{
272272
"metadata": {},
@@ -291,7 +291,7 @@
291291
" log_level=\"INFO\",\n",
292292
" )"
293293
],
294-
"id": "dbfb32f2a0f7b8ef"
294+
"id": "cb04960f17cc9b75"
295295
},
296296
{
297297
"metadata": {},
@@ -324,7 +324,7 @@
324324
"\n",
325325
"This method is not recommended, as zFPKM is much more robust for a similar level of \"hands-off\" model building\n"
326326
],
327-
"id": "a820a3985a2601c7"
327+
"id": "4bd864328c93b27a"
328328
},
329329
{
330330
"metadata": {},
@@ -345,7 +345,7 @@
345345
"#### Single Cell RNA Sequencing\n",
346346
"While the Snakemake pipeline does not yet support single-cell alignment, and COMO does not yet support automated configuration file and counts matrix file creation for single-cell alignment output from STAR, it is possible to use single-cell data to create a model with COMO. Because normalization strategies can be applied to single-cell data in the same way it is applied to bulk RNA sequencing, `como/rnaseq_gen.py` can be used with a provided counts matrix and configuration file, from [Step 1](Step-1:-Initialize-and-Preprocess-RNA-seq-data), above. Just like `\"total\"` and `\"mRNA\"`, `como/rnaseq_gen.py` can be executed with `\"SC\"` as the \"`--library-prep`\" argument to help COMO differentiate it from any bulk RNA sequencing data if multiple strategies are being used."
347347
],
348-
"id": "a1d6608d045c7399"
348+
"id": "e99e612919298aa8"
349349
},
350350
{
351351
"metadata": {},
@@ -364,7 +364,7 @@
364364
"- `min_zfpkm`: The cutoff for Counts-Per-Million filtering\n",
365365
"- `prep_method`: The library method used for preparation. Options are: `\"total\"`, `\"mRNA\"`, or `\"SC\"`,\n"
366366
],
367-
"id": "1fcf38e9cbc3de48"
367+
"id": "8e550961a4a443b3"
368368
},
369369
{
370370
"metadata": {},
@@ -400,7 +400,7 @@
400400
" cutoff=cutoff\n",
401401
" )"
402402
],
403-
"id": "5a2e17b646d6de36"
403+
"id": "d98c023e7738eb64"
404404
},
405405
{
406406
"metadata": {},
@@ -420,7 +420,7 @@
420420
"- `min_zfpkm`: The cutoff for Counts-Per-Million filtering\n",
421421
"- `prep_method`: The library method used for preparation. Options are: `\"total\"`, `\"mRNA\"`, or `\"SC\"`,\n"
422422
],
423-
"id": "87b5397301541650"
423+
"id": "28848c3d315fa78c"
424424
},
425425
{
426426
"metadata": {},
@@ -456,7 +456,7 @@
456456
" cutoff=cutoff\n",
457457
" )"
458458
],
459-
"id": "3e73c07fa4a9e49a"
459+
"id": "3e87f9890bc76e43"
460460
},
461461
{
462462
"metadata": {},
@@ -476,7 +476,7 @@
476476
"- `min_zfpkm`: The cutoff for Counts-Per-Million filtering\n",
477477
"- `prep_method`: The library method used for preparation. Options are: `\"total\"`, `\"mRNA\"`, or `\"scrna\"`,\n"
478478
],
479-
"id": "ae1278480199f415"
479+
"id": "bfa9e2cd4152f961"
480480
},
481481
{
482482
"metadata": {},
@@ -512,7 +512,7 @@
512512
" cutoff=cutoff\n",
513513
" )"
514514
],
515-
"id": "567734d827571d6f"
515+
"id": "c45085c6c12d9a3c"
516516
},
517517
{
518518
"metadata": {},
@@ -529,7 +529,7 @@
529529
"- `high_batch_ratio`: The ratio required before a gene is considered \"high-confidence\" in the study\n",
530530
"- `quantile`: The cutoff Transcripts-Per-Million quantile for filtering"
531531
],
532-
"id": "ea7004d23807dd1"
532+
"id": "f01f21670b1e6956"
533533
},
534534
{
535535
"metadata": {},
@@ -556,7 +556,7 @@
556556
" quantile=25,\n",
557557
" )"
558558
],
559-
"id": "a7715d2cf9b0191a"
559+
"id": "dcbae45189dd2eba"
560560
},
561561
{
562562
"metadata": {},
@@ -582,7 +582,7 @@
582582
"- `n_neighbors_context`: N nearest neighbors for context clustering. The default is `\"default\"`, which is the total number of contexts\n",
583583
"- `seed`: The random seed for clustering algorithm initialization. If not specified, `np.random.randint(0, 100000)` is used"
584584
],
585-
"id": "f4f395e25eeb1185"
585+
"id": "354abde9f52ff9cf"
586586
},
587587
{
588588
"metadata": {},
@@ -626,7 +626,7 @@
626626
"\n",
627627
"!{cmd}"
628628
],
629-
"id": "cc61326b8df01fc4"
629+
"id": "be527c98a8523323"
630630
},
631631
{
632632
"metadata": {},
@@ -666,7 +666,7 @@
666666
"\n",
667667
"Each of the \"weights\" (`total_rna_weight`, `mrna_weight`, etc.) are used to place a significance on each method. Becuase there are many steps in the Dogma from transcription to translation, the gene expression as seen by total RNA or mRNA sequencing may not be representative of the gene's protein expression, and this its metabolic impact. Because of this, you are able to weight each source more (or less) than another."
668668
],
669-
"id": "236fb36fa41f60b"
669+
"id": "db0cfbec103d7d35"
670670
},
671671
{
672672
"metadata": {},
@@ -703,7 +703,7 @@
703703
"\n",
704704
"!{cmd}"
705705
],
706-
"id": "3ce78fc277c55885"
706+
"id": "75b84df4d88d7cbe"
707707
},
708708
{
709709
"metadata": {},
@@ -770,7 +770,7 @@
770770
"- `force_reactions_filename`: The filename of the force reactions to be used. Force reactions will (as the name implies) force the optimizer to use these reactions, **no matter their expression**\n",
771771
"- `exclude_reactions_filename`: The filename of reactions to exclude from the model, no matter their expression"
772772
],
773-
"id": "b3a182617ff3eb7a"
773+
"id": "30da70d405a80d23"
774774
},
775775
{
776776
"metadata": {},
@@ -839,7 +839,7 @@
839839
" # fmt: on\n",
840840
" !{cmd}"
841841
],
842-
"id": "e94c36e65ba46e38"
842+
"id": "19918f482629d4fe"
843843
},
844844
{
845845
"metadata": {},
@@ -864,7 +864,7 @@
864864
"- `exampleTissue`: This is the name of the tissue context\n",
865865
"- `ALGORITHM`: This is the algorithm (`recon_algorithm`) used in the above model creation step\n"
866866
],
867-
"id": "47e3e576b43a040d"
867+
"id": "38640adc60fe1909"
868868
},
869869
{
870870
"metadata": {},
@@ -941,7 +941,7 @@
941941
"\n",
942942
" !{cmd}"
943943
],
944-
"id": "4a064f3322a3991e"
944+
"id": "ef639305627a4a4"
945945
},
946946
{
947947
"metadata": {},
@@ -960,7 +960,7 @@
960960
"- `data_source`: The datasource you are using for disease analysis. This should be`\"rnaseq\"`\n",
961961
"- `taxon_id`: The [NCBI Taxon ID](https://www.ncbi.nlm.nih.gov/taxonomy) to use for disease analysis"
962962
],
963-
"id": "3aa49b19ab4fc4b9"
963+
"id": "ad06b9e81d3d5f39"
964964
},
965965
{
966966
"metadata": {},
@@ -992,7 +992,7 @@
992992
"\n",
993993
" !{cmd}"
994994
],
995-
"id": "7f946d38cac3a18c"
995+
"id": "c9d0e6dfd35adefb"
996996
},
997997
{
998998
"metadata": {},
@@ -1027,7 +1027,7 @@
10271027
"\n",
10281028
"- `solver`: The solver you would like to use. Available options are `\"gurobi\"` or `\"glpk\"`\n"
10291029
],
1030-
"id": "feb9da60acb9a3b3"
1030+
"id": "7093fbd5a6551c18"
10311031
},
10321032
{
10331033
"metadata": {},
@@ -1099,7 +1099,7 @@
10991099
" cmd = \" \".join(cmd)\n",
11001100
" !{cmd}"
11011101
],
1102-
"id": "c999b3f953381f9f"
1102+
"id": "281a24eba5e06fc9"
11031103
}
11041104
],
11051105
"metadata": {},

main/como/cluster_rnaseq.py

Lines changed: 3 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -57,19 +57,13 @@ def __post_init__(self): # noqa: C901, ignore too complex
5757
if 0 > self.min_distance > 1.0:
5858
raise ValueError("--min_dist must be a float between 0 and 1")
5959

60-
if (
61-
isdigit(self.num_replicate_neighbors) and self.num_replicate_neighbors < 1
62-
) or self.num_replicate_neighbors != "default":
60+
if (isdigit(self.num_replicate_neighbors) and self.num_replicate_neighbors < 1) or self.num_replicate_neighbors != "default":
6361
raise ValueError("--n-neighbors-rep must be either 'default' or an integer > 1")
6462

65-
if (
66-
isdigit(self.num_batch_neighbors) and self.num_batch_neighbors < 1
67-
) or self.num_batch_neighbors != "default":
63+
if (isdigit(self.num_batch_neighbors) and self.num_batch_neighbors < 1) or self.num_batch_neighbors != "default":
6864
raise ValueError("--n-neighbors-batch must be either 'default' or an integer > 1")
6965

70-
if (
71-
isdigit(self.num_context_neighbors) and self.num_context_neighbors < 1
72-
) or self.num_context_neighbors != "default":
66+
if (isdigit(self.num_context_neighbors) and self.num_context_neighbors < 1) or self.num_context_neighbors != "default":
7367
raise ValueError("--n-neighbors-context must be either 'default' or an integer > 1")
7468

7569

main/como/combine_distributions.py

Lines changed: 10 additions & 44 deletions
Original file line numberDiff line numberDiff line change
@@ -59,12 +59,7 @@ def _merge_batch(wd, context, batch):
5959
zmat.columns = rep_names
6060

6161
stack_df = pd.concat(
62-
[
63-
pd.DataFrame(
64-
{"entrez_gene_id": zmat["entrez_gene_id"], "zscore": zmat[col].astype(float), "source": col}
65-
)
66-
for col in zmat.columns[1:]
67-
]
62+
[pd.DataFrame({"entrez_gene_id": zmat["entrez_gene_id"], "zscore": zmat[col].astype(float), "source": col}) for col in zmat.columns[1:]]
6863
)
6964

7065
plot_name_png = wd / "figures" / f"plot_{context}_{Path(f).stem}.png"
@@ -108,9 +103,7 @@ def weighted_z(x):
108103

109104
stack_df = pd.concat(
110105
[
111-
pd.DataFrame(
112-
{"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col}
113-
)
106+
pd.DataFrame({"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col})
114107
for col in merge_df.columns[1:]
115108
]
116109
)
@@ -157,9 +150,7 @@ def weighted_z(x, n_reps):
157150

158151
stack_df = pd.concat(
159152
[
160-
pd.DataFrame(
161-
{"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col}
162-
)
153+
pd.DataFrame({"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col})
163154
for col in merge_df.columns[1:]
164155
]
165156
)
@@ -259,9 +250,7 @@ def weighted_z(x, weights):
259250

260251
stack_df = pd.concat(
261252
[
262-
pd.DataFrame(
263-
{"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col}
264-
)
253+
pd.DataFrame({"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col})
265254
for col in merge_df.columns[1:]
266255
]
267256
)
@@ -343,11 +332,7 @@ def _combine_zscores( # noqa: C901
343332
num_reps.extend(res[1])
344333
combine_z_matrix = _combine_batch_zdistro(trna_workdir, context, batch, z_matrix)
345334
combine_z_matrix.columns = ["entrez_gene_id", batch]
346-
merge_z_data = (
347-
combine_z_matrix
348-
if merge_z_data.empty
349-
else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer")
350-
)
335+
merge_z_data = combine_z_matrix if merge_z_data.empty else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer")
351336

352337
comb_batches_z_trna = _combine_context_zdistro(trna_workdir, context, num_reps, merge_z_data)
353338
filename = trna_workdir / f"combined_zFPKM_{context}.csv"
@@ -369,11 +354,7 @@ def _combine_zscores( # noqa: C901
369354
num_reps.extend(res[1])
370355
combine_z_matrix = _combine_batch_zdistro(mrna_workdir, context, batch, z_matrix)
371356
combine_z_matrix.columns = ["entrez_gene_id", batch]
372-
merge_z_data = (
373-
combine_z_matrix
374-
if merge_z_data.empty
375-
else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer")
376-
)
357+
merge_z_data = combine_z_matrix if merge_z_data.empty else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer")
377358

378359
comb_batches_z_mrna = _combine_context_zdistro(mrna_workdir, context, num_reps, merge_z_data)
379360
filename = mrna_workdir / f"combined_zFPKM_{context}.csv"
@@ -395,11 +376,7 @@ def _combine_zscores( # noqa: C901
395376
num_reps.extend(res[1])
396377
combine_z_matrix = _combine_batch_zdistro(scrna_workdir, context, batch, z_matrix)
397378
combine_z_matrix.columns = ["entrez_gene_id", batch]
398-
merge_z_data = (
399-
combine_z_matrix
400-
if merge_z_data.empty
401-
else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer")
402-
)
379+
merge_z_data = combine_z_matrix if merge_z_data.empty else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer")
403380

404381
comb_batches_z_scrna = _combine_context_zdistro(scrna_workdir, context, num_reps, merge_z_data)
405382
filename = scrna_workdir / f"combined_zFPKM_{context}.csv"
@@ -421,11 +398,7 @@ def _combine_zscores( # noqa: C901
421398
num_reps.extend(res[1])
422399
combine_z_matrix = _combine_batch_zdistro(protein_workdir, context, batch, z_matrix)
423400
combine_z_matrix.columns = ["entrez_gene_id", batch]
424-
merge_z_data = (
425-
combine_z_matrix
426-
if merge_z_data.empty
427-
else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer")
428-
)
401+
merge_z_data = combine_z_matrix if merge_z_data.empty else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer")
429402

430403
comb_batches_z_prot = _combine_context_zdistro(protein_workdir, context, num_reps, merge_z_data)
431404
filename = protein_workdir / f"combined_zscore_proteinAbundance_{context}.csv"
@@ -435,15 +408,8 @@ def _combine_zscores( # noqa: C901
435408
filename = protein_workdir / f"model_scores_{context}.csv"
436409
comb_batches_z_prot.to_csv(filename, index=False)
437410

438-
if (
439-
comb_batches_z_mrna is None
440-
and comb_batches_z_trna is None
441-
and comb_batches_z_scrna is None
442-
and comb_batches_z_prot is None
443-
):
444-
logger.critical(
445-
f"The context '{context}' was found in the configuration file but no data was found on disk!"
446-
)
411+
if comb_batches_z_mrna is None and comb_batches_z_trna is None and comb_batches_z_scrna is None and comb_batches_z_prot is None:
412+
logger.critical(f"The context '{context}' was found in the configuration file but no data was found on disk!")
447413
continue
448414

449415
comb_omics_z = _combine_omics_zdistros(

0 commit comments

Comments
 (0)