diff --git a/main/COMO.ipynb b/main/COMO.ipynb index 6b5b7f51..5bc09cb6 100644 --- a/main/COMO.ipynb +++ b/main/COMO.ipynb @@ -53,7 +53,7 @@ " 3. Compare results between perturbed and unperturbed models (i.e., knocked-out models vs non-knocked-out models)\n", " 4. Integrate with disease genes and create a score of drug targets" ], - "id": "ea1c9614518c30bc" + "id": "90fba8b56a999517" }, { "metadata": {}, @@ -223,7 +223,7 @@ " \n", "" ], - "id": "28d59417491e63db" + "id": "dcc8d38caab63fab" }, { "metadata": {}, @@ -254,7 +254,7 @@ "scrna_matrix_filepath = {context: Path(f\"data/results/{context}/scrna/scrna_{context}.csv\") for context in context_names}\n", "# scrna_matrix_filepath = [Path(f\"data/results/{context}/scrna/scrna_{context}.h5ad\") for context in context_names]\n" ], - "id": "18061578964ca096" + "id": "fd4c387eab48e9ee" }, { "metadata": {}, @@ -266,7 +266,7 @@ "- `taxon_id`: The [NCBI Taxon ID](https://www.ncbi.nlm.nih.gov/taxonomy) to use\n", "- `preprocess_mode`: This should be set to `\"create-matrix\"` if you are **not** providing a matrix, otherwise set it to `\"provide-matrix\"`" ], - "id": "b475e41b000d0cb5" + "id": "24a8475316ff6d70" }, { "metadata": {}, @@ -291,7 +291,7 @@ " log_level=\"INFO\",\n", " )" ], - "id": "dbfb32f2a0f7b8ef" + "id": "cb04960f17cc9b75" }, { "metadata": {}, @@ -324,7 +324,7 @@ "\n", "This method is not recommended, as zFPKM is much more robust for a similar level of \"hands-off\" model building\n" ], - "id": "a820a3985a2601c7" + "id": "4bd864328c93b27a" }, { "metadata": {}, @@ -345,7 +345,7 @@ "#### Single Cell RNA Sequencing\n", "While the Snakemake pipeline does not yet support single-cell alignment, and COMO does not yet support automated configuration file and counts matrix file creation for single-cell alignment output from STAR, it is possible to use single-cell data to create a model with COMO. Because normalization strategies can be applied to single-cell data in the same way it is applied to bulk RNA sequencing, `como/rnaseq_gen.py` can be used with a provided counts matrix and configuration file, from [Step 1](Step-1:-Initialize-and-Preprocess-RNA-seq-data), above. Just like `\"total\"` and `\"mRNA\"`, `como/rnaseq_gen.py` can be executed with `\"SC\"` as the \"`--library-prep`\" argument to help COMO differentiate it from any bulk RNA sequencing data if multiple strategies are being used." ], - "id": "a1d6608d045c7399" + "id": "e99e612919298aa8" }, { "metadata": {}, @@ -364,7 +364,7 @@ "- `min_zfpkm`: The cutoff for Counts-Per-Million filtering\n", "- `prep_method`: The library method used for preparation. Options are: `\"total\"`, `\"mRNA\"`, or `\"SC\"`,\n" ], - "id": "1fcf38e9cbc3de48" + "id": "8e550961a4a443b3" }, { "metadata": {}, @@ -400,7 +400,7 @@ " cutoff=cutoff\n", " )" ], - "id": "5a2e17b646d6de36" + "id": "d98c023e7738eb64" }, { "metadata": {}, @@ -420,7 +420,7 @@ "- `min_zfpkm`: The cutoff for Counts-Per-Million filtering\n", "- `prep_method`: The library method used for preparation. Options are: `\"total\"`, `\"mRNA\"`, or `\"SC\"`,\n" ], - "id": "87b5397301541650" + "id": "28848c3d315fa78c" }, { "metadata": {}, @@ -456,7 +456,7 @@ " cutoff=cutoff\n", " )" ], - "id": "3e73c07fa4a9e49a" + "id": "3e87f9890bc76e43" }, { "metadata": {}, @@ -476,7 +476,7 @@ "- `min_zfpkm`: The cutoff for Counts-Per-Million filtering\n", "- `prep_method`: The library method used for preparation. Options are: `\"total\"`, `\"mRNA\"`, or `\"scrna\"`,\n" ], - "id": "ae1278480199f415" + "id": "bfa9e2cd4152f961" }, { "metadata": {}, @@ -512,7 +512,7 @@ " cutoff=cutoff\n", " )" ], - "id": "567734d827571d6f" + "id": "c45085c6c12d9a3c" }, { "metadata": {}, @@ -529,7 +529,7 @@ "- `high_batch_ratio`: The ratio required before a gene is considered \"high-confidence\" in the study\n", "- `quantile`: The cutoff Transcripts-Per-Million quantile for filtering" ], - "id": "ea7004d23807dd1" + "id": "f01f21670b1e6956" }, { "metadata": {}, @@ -556,7 +556,7 @@ " quantile=25,\n", " )" ], - "id": "a7715d2cf9b0191a" + "id": "dcbae45189dd2eba" }, { "metadata": {}, @@ -582,7 +582,7 @@ "- `n_neighbors_context`: N nearest neighbors for context clustering. The default is `\"default\"`, which is the total number of contexts\n", "- `seed`: The random seed for clustering algorithm initialization. If not specified, `np.random.randint(0, 100000)` is used" ], - "id": "f4f395e25eeb1185" + "id": "354abde9f52ff9cf" }, { "metadata": {}, @@ -626,7 +626,7 @@ "\n", "!{cmd}" ], - "id": "cc61326b8df01fc4" + "id": "be527c98a8523323" }, { "metadata": {}, @@ -666,7 +666,7 @@ "\n", "Each of the \"weights\" (`total_rna_weight`, `mrna_weight`, etc.) are used to place a significance on each method. Becuase there are many steps in the Dogma from transcription to translation, the gene expression as seen by total RNA or mRNA sequencing may not be representative of the gene's protein expression, and this its metabolic impact. Because of this, you are able to weight each source more (or less) than another." ], - "id": "236fb36fa41f60b" + "id": "db0cfbec103d7d35" }, { "metadata": {}, @@ -703,7 +703,7 @@ "\n", "!{cmd}" ], - "id": "3ce78fc277c55885" + "id": "75b84df4d88d7cbe" }, { "metadata": {}, @@ -770,7 +770,7 @@ "- `force_reactions_filename`: The filename of the force reactions to be used. Force reactions will (as the name implies) force the optimizer to use these reactions, **no matter their expression**\n", "- `exclude_reactions_filename`: The filename of reactions to exclude from the model, no matter their expression" ], - "id": "b3a182617ff3eb7a" + "id": "30da70d405a80d23" }, { "metadata": {}, @@ -839,7 +839,7 @@ " # fmt: on\n", " !{cmd}" ], - "id": "e94c36e65ba46e38" + "id": "19918f482629d4fe" }, { "metadata": {}, @@ -864,7 +864,7 @@ "- `exampleTissue`: This is the name of the tissue context\n", "- `ALGORITHM`: This is the algorithm (`recon_algorithm`) used in the above model creation step\n" ], - "id": "47e3e576b43a040d" + "id": "38640adc60fe1909" }, { "metadata": {}, @@ -941,7 +941,7 @@ "\n", " !{cmd}" ], - "id": "4a064f3322a3991e" + "id": "ef639305627a4a4" }, { "metadata": {}, @@ -960,7 +960,7 @@ "- `data_source`: The datasource you are using for disease analysis. This should be`\"rnaseq\"`\n", "- `taxon_id`: The [NCBI Taxon ID](https://www.ncbi.nlm.nih.gov/taxonomy) to use for disease analysis" ], - "id": "3aa49b19ab4fc4b9" + "id": "ad06b9e81d3d5f39" }, { "metadata": {}, @@ -992,7 +992,7 @@ "\n", " !{cmd}" ], - "id": "7f946d38cac3a18c" + "id": "c9d0e6dfd35adefb" }, { "metadata": {}, @@ -1027,7 +1027,7 @@ "\n", "- `solver`: The solver you would like to use. Available options are `\"gurobi\"` or `\"glpk\"`\n" ], - "id": "feb9da60acb9a3b3" + "id": "7093fbd5a6551c18" }, { "metadata": {}, @@ -1099,7 +1099,7 @@ " cmd = \" \".join(cmd)\n", " !{cmd}" ], - "id": "c999b3f953381f9f" + "id": "281a24eba5e06fc9" } ], "metadata": {}, diff --git a/main/como/cluster_rnaseq.py b/main/como/cluster_rnaseq.py index dba5499c..fc3a9f48 100644 --- a/main/como/cluster_rnaseq.py +++ b/main/como/cluster_rnaseq.py @@ -57,19 +57,13 @@ def __post_init__(self): # noqa: C901, ignore too complex if 0 > self.min_distance > 1.0: raise ValueError("--min_dist must be a float between 0 and 1") - if ( - isdigit(self.num_replicate_neighbors) and self.num_replicate_neighbors < 1 - ) or self.num_replicate_neighbors != "default": + if (isdigit(self.num_replicate_neighbors) and self.num_replicate_neighbors < 1) or self.num_replicate_neighbors != "default": raise ValueError("--n-neighbors-rep must be either 'default' or an integer > 1") - if ( - isdigit(self.num_batch_neighbors) and self.num_batch_neighbors < 1 - ) or self.num_batch_neighbors != "default": + if (isdigit(self.num_batch_neighbors) and self.num_batch_neighbors < 1) or self.num_batch_neighbors != "default": raise ValueError("--n-neighbors-batch must be either 'default' or an integer > 1") - if ( - isdigit(self.num_context_neighbors) and self.num_context_neighbors < 1 - ) or self.num_context_neighbors != "default": + if (isdigit(self.num_context_neighbors) and self.num_context_neighbors < 1) or self.num_context_neighbors != "default": raise ValueError("--n-neighbors-context must be either 'default' or an integer > 1") diff --git a/main/como/combine_distributions.py b/main/como/combine_distributions.py index ed69413b..68939dda 100644 --- a/main/como/combine_distributions.py +++ b/main/como/combine_distributions.py @@ -59,12 +59,7 @@ def _merge_batch(wd, context, batch): zmat.columns = rep_names stack_df = pd.concat( - [ - pd.DataFrame( - {"entrez_gene_id": zmat["entrez_gene_id"], "zscore": zmat[col].astype(float), "source": col} - ) - for col in zmat.columns[1:] - ] + [pd.DataFrame({"entrez_gene_id": zmat["entrez_gene_id"], "zscore": zmat[col].astype(float), "source": col}) for col in zmat.columns[1:]] ) plot_name_png = wd / "figures" / f"plot_{context}_{Path(f).stem}.png" @@ -108,9 +103,7 @@ def weighted_z(x): stack_df = pd.concat( [ - pd.DataFrame( - {"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col} - ) + pd.DataFrame({"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col}) for col in merge_df.columns[1:] ] ) @@ -157,9 +150,7 @@ def weighted_z(x, n_reps): stack_df = pd.concat( [ - pd.DataFrame( - {"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col} - ) + pd.DataFrame({"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col}) for col in merge_df.columns[1:] ] ) @@ -259,9 +250,7 @@ def weighted_z(x, weights): stack_df = pd.concat( [ - pd.DataFrame( - {"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col} - ) + pd.DataFrame({"entrez_gene_id": merge_df["entrez_gene_id"], "zscore": merge_df[col].astype(float), "source": col}) for col in merge_df.columns[1:] ] ) @@ -343,11 +332,7 @@ def _combine_zscores( # noqa: C901 num_reps.extend(res[1]) combine_z_matrix = _combine_batch_zdistro(trna_workdir, context, batch, z_matrix) combine_z_matrix.columns = ["entrez_gene_id", batch] - merge_z_data = ( - combine_z_matrix - if merge_z_data.empty - else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer") - ) + merge_z_data = combine_z_matrix if merge_z_data.empty else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer") comb_batches_z_trna = _combine_context_zdistro(trna_workdir, context, num_reps, merge_z_data) filename = trna_workdir / f"combined_zFPKM_{context}.csv" @@ -369,11 +354,7 @@ def _combine_zscores( # noqa: C901 num_reps.extend(res[1]) combine_z_matrix = _combine_batch_zdistro(mrna_workdir, context, batch, z_matrix) combine_z_matrix.columns = ["entrez_gene_id", batch] - merge_z_data = ( - combine_z_matrix - if merge_z_data.empty - else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer") - ) + merge_z_data = combine_z_matrix if merge_z_data.empty else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer") comb_batches_z_mrna = _combine_context_zdistro(mrna_workdir, context, num_reps, merge_z_data) filename = mrna_workdir / f"combined_zFPKM_{context}.csv" @@ -395,11 +376,7 @@ def _combine_zscores( # noqa: C901 num_reps.extend(res[1]) combine_z_matrix = _combine_batch_zdistro(scrna_workdir, context, batch, z_matrix) combine_z_matrix.columns = ["entrez_gene_id", batch] - merge_z_data = ( - combine_z_matrix - if merge_z_data.empty - else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer") - ) + merge_z_data = combine_z_matrix if merge_z_data.empty else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer") comb_batches_z_scrna = _combine_context_zdistro(scrna_workdir, context, num_reps, merge_z_data) filename = scrna_workdir / f"combined_zFPKM_{context}.csv" @@ -421,11 +398,7 @@ def _combine_zscores( # noqa: C901 num_reps.extend(res[1]) combine_z_matrix = _combine_batch_zdistro(protein_workdir, context, batch, z_matrix) combine_z_matrix.columns = ["entrez_gene_id", batch] - merge_z_data = ( - combine_z_matrix - if merge_z_data.empty - else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer") - ) + merge_z_data = combine_z_matrix if merge_z_data.empty else pd.merge(merge_z_data, combine_z_matrix, on="entrez_gene_id", how="outer") comb_batches_z_prot = _combine_context_zdistro(protein_workdir, context, num_reps, merge_z_data) filename = protein_workdir / f"combined_zscore_proteinAbundance_{context}.csv" @@ -435,15 +408,8 @@ def _combine_zscores( # noqa: C901 filename = protein_workdir / f"model_scores_{context}.csv" comb_batches_z_prot.to_csv(filename, index=False) - if ( - comb_batches_z_mrna is None - and comb_batches_z_trna is None - and comb_batches_z_scrna is None - and comb_batches_z_prot is None - ): - logger.critical( - f"The context '{context}' was found in the configuration file but no data was found on disk!" - ) + if comb_batches_z_mrna is None and comb_batches_z_trna is None and comb_batches_z_scrna is None and comb_batches_z_prot is None: + logger.critical(f"The context '{context}' was found in the configuration file but no data was found on disk!") continue comb_omics_z = _combine_omics_zdistros( diff --git a/main/como/create_context_specific_model.py b/main/como/create_context_specific_model.py index 4e483985..f6f71fa5 100644 --- a/main/como/create_context_specific_model.py +++ b/main/como/create_context_specific_model.py @@ -86,15 +86,9 @@ class _Arguments: def __post_init__(self): self.reference_model = Path(self.reference_model) self.active_genes_filepath = Path(self.active_genes_filepath) - self.boundary_reactions_filepath = ( - Path(self.boundary_reactions_filepath) if self.boundary_reactions_filepath is not None else None - ) - self.exclude_reactions_filepath = ( - Path(self.exclude_reactions_filepath) if self.exclude_reactions_filepath is not None else None - ) - self.force_reactions_filepath = ( - Path(self.force_reactions_filepath) if self.force_reactions_filepath is not None else None - ) + self.boundary_reactions_filepath = Path(self.boundary_reactions_filepath) if self.boundary_reactions_filepath is not None else None + self.exclude_reactions_filepath = Path(self.exclude_reactions_filepath) if self.exclude_reactions_filepath is not None else None + self.force_reactions_filepath = Path(self.force_reactions_filepath) if self.force_reactions_filepath is not None else None if not self.reference_model.exists(): raise FileNotFoundError(f"Reference model not found at {self.reference_model}") @@ -109,8 +103,7 @@ def __post_init__(self): if self.high_threshold < self.low_threshold: raise ValueError( - f"Low threshold must be less than high threshold. " - f"Received low threshold: {self.low_threshold}, high threshold: {self.high_threshold}" + f"Low threshold must be less than high threshold. Received low threshold: {self.low_threshold}, high threshold: {self.high_threshold}" ) @@ -283,12 +276,9 @@ def _build_with_fastcore(cobra_model, s_matrix, lb, ub, exp_idx_list, solver): # 'Vlassis, Pacheco, Sauter (2014). Fast reconstruction of compact # context-specific metabolic network models. PLoS Comput. Biol. 10, # e1003424.' - logger.warning( - "Fastcore requires a flux consistant model is used as refererence, " - "to achieve this fastcc is required which is NOT reproducible." - ) + logger.warning("Fastcore requires a flux consistant model is used as refererence, to achieve this fastcc is required which is NOT reproducible.") logger.debug("Creating feasible model") - incon_rxns, cobra_model = _feasibility_test(cobra_model, "other") + _, cobra_model = _feasibility_test(cobra_model, "other") properties = FastcoreProperties(core=exp_idx_list, solver=solver) algorithm = FASTcore(s_matrix, lb, ub, properties) context_rxns = algorithm.fastcore() @@ -383,10 +373,7 @@ def _map_expression_to_reaction( if gene_reaction_rule.strip() == "": continue for gid in gene_ids: - if gid in gene_expressions.index: - rep_val = f" {gene_expressions.at[gid, 'active']} " - else: - rep_val = f" {unknown_val!s} " + rep_val = f" {gene_expressions.at[gid, 'active']} " if gid in gene_expressions.index else f" {unknown_val!s} " gene_reaction_rule = f" {gene_reaction_rule} " # pad white space to prevent gene matches inside floats gene_reaction_rule = gene_reaction_rule.replace(f" {gid} ", rep_val, 1) try: @@ -435,9 +422,7 @@ def _build_model( # noqa: C901 case ".json": reference_model = cobra.io.load_json_model(general_model_file) case _: - raise NameError( - f"Reference reference_model format must be .xml, .mat, or .json, found '{general_model_file.suffix}'" - ) + raise NameError(f"Reference reference_model format must be .xml, .mat, or .json, found '{general_model_file.suffix}'") reference_model.objective = {getattr(reference_model.reactions, objective): 1} # set objective @@ -445,7 +430,7 @@ def _build_model( # noqa: C901 force_rxns.append(objective) # set boundaries - reference_model, bound_rm_rxns = _set_boundaries(reference_model, bound_rxns, bound_lb, bound_ub) + reference_model, _ = _set_boundaries(reference_model, bound_rxns, bound_lb, bound_ub) # set solver reference_model.solver = solver.lower() @@ -681,14 +666,10 @@ def create_context_specific_model( # noqa: C901 ) config = Config() - build_results.infeasible_reactions.to_csv( - config.result_dir / context_name / f"{context_name}_infeasible_rxns.csv", index=False - ) + build_results.infeasible_reactions.to_csv(config.result_dir / context_name / f"{context_name}_infeasible_rxns.csv", index=False) if algorithm == Algorithm.FASTCORE: - pd.DataFrame(build_results.expression_index_list).to_csv( - config.result_dir / context_name / f"{context_name}_core_rxns.csv", index=False - ) + pd.DataFrame(build_results.expression_index_list).to_csv(config.result_dir / context_name / f"{context_name}_core_rxns.csv", index=False) output_directory = config.result_dir / context_name _write_model_to_disk( @@ -789,8 +770,7 @@ def _parse_args(): type=str, default="GIMME", dest="recon_algorithm", - help="Algorithm used to seed context specific model to the Genome-scale model. " - "Can be either GIMME, FASTCORE, iMAT, or tINIT.", + help="Algorithm used to seed context specific model to the Genome-scale model. Can be either GIMME, FASTCORE, iMAT, or tINIT.", ) parser.add_argument( "-lt", diff --git a/main/como/knock_out_simulation.py b/main/como/knock_out_simulation.py index fe96c784..936996c7 100644 --- a/main/como/knock_out_simulation.py +++ b/main/como/knock_out_simulation.py @@ -280,8 +280,7 @@ def main(argv): parser = argparse.ArgumentParser( prog="knock_out_simulation.py", description="This script is responsible for mapping drug targets in metabolic models, performing knock out simulations, and comparing simulation results with disease genes. It also identified drug targets and repurposable drugs.", - epilog="For additional help, please post questions/issues in the MADRID GitHub repo at " - "https://github.com/HelikarLab/COMO", + epilog="For additional help, please post questions/issues in the MADRID GitHub repo at https://github.com/HelikarLab/COMO", ) parser.add_argument( "-m", diff --git a/main/como/merge_xomics.py b/main/como/merge_xomics.py index 77d2336a..e6c996e2 100644 --- a/main/como/merge_xomics.py +++ b/main/como/merge_xomics.py @@ -80,14 +80,10 @@ def __post_init__(self): if self.expression_requirement.isdigit(): self.expression_requirement = int(self.expression_requirement) if self.expression_requirement < 1: - logger.warning( - f"Expression requirement should be at least 1, setting to 1 now. Got {self.expression_requirement}" - ) + logger.warning(f"Expression requirement should be at least 1, setting to 1 now. Got {self.expression_requirement}") self.expression_requirement = 1 elif self.expression_requirement != "default": - raise ValueError( - f"Expression requirement should be an integer or 'default', got {self.expression_requirement}" - ) + raise ValueError(f"Expression requirement should be an integer or 'default', got {self.expression_requirement}") if self.adjustment_method.value.lower() not in ["progressive", "regressive", "flat", "custom"]: raise ValueError("Adjust method must be either 'progressive', 'regressive', 'flat', or 'custom'") @@ -119,10 +115,7 @@ def load_dummy_dict(): case RNAPrepMethod.SCRNA: filename = f"rnaseq_scrna_{context_name}.csv" case _: - raise ValueError( - f"Unsupported RNA-seq library type: {prep_method.value}. " - f"Must be an option defined in 'RNASeqPreparationMethod'." - ) + raise ValueError(f"Unsupported RNA-seq library type: {prep_method.value}. Must be an option defined in 'RNASeqPreparationMethod'.") save_filepath = config.result_dir / context_name / prep_method.value / filename if save_filepath.exists(): @@ -150,9 +143,7 @@ def _merge_logical_table(df: pd.DataFrame): df.dropna(axis=0, subset=["entrez_gene_id"], inplace=True) df["entrez_gene_id"] = df["entrez_gene_id"].astype(str).str.replace(" /// ", "//").astype(str) - id_list: list[str] = df.loc[ - ~df["entrez_gene_id"].str.contains("//"), "entrez_gene_id" - ].tolist() # Collect "single" ids, like "123" + id_list: list[str] = df.loc[~df["entrez_gene_id"].str.contains("//"), "entrez_gene_id"].tolist() # Collect "single" ids, like "123" multiple_entrez_ids: list[str] = df.loc[ df["entrez_gene_id"].str.contains("//"), "entrez_gene_id" ].tolist() # Collect "double" ids, like "123//456" @@ -285,8 +276,7 @@ async def _get_transcriptmoic_details(merged_df: pd.DataFrame) -> pd.DataFrame: ] gene_details["gene_info_type"] = [ - i.group(1) if isinstance(i, re.Match) else "None" - for i in gene_details["Gene Info"].apply(lambda x: re.search(r"\[Gene Type: (.*)\]", x)) + i.group(1) if isinstance(i, re.Match) else "None" for i in gene_details["Gene Info"].apply(lambda x: re.search(r"\[Gene Type: (.*)\]", x)) ] gene_details["ensembl_info_type"] = [ i.group(1) if isinstance(i, re.Match) else "None" @@ -406,9 +396,7 @@ async def _merge_xomics( merge_data = prote_data if merge_data is None else merge_data.join(prote_data, how="outer") if merge_data is None: - logger.critical( - f"No data is available for the '{context_name}' context. If this is intentional, ignore this error." - ) + logger.critical(f"No data is available for the '{context_name}' context. If this is intentional, ignore this error.") return {} merge_data = _merge_logical_table(merge_data) @@ -423,9 +411,7 @@ async def _merge_xomics( ) else: # subtract one from requirement per NA merge_data.loc[:, "Required"] = merge_data[expression_list].apply( - lambda x: expression_requirement - (num_sources - x.count()) - if (expression_requirement - (num_sources - x.count()) > 0) - else 1, + lambda x: expression_requirement - (num_sources - x.count()) if (expression_requirement - (num_sources - x.count()) > 0) else 1, axis=1, ) @@ -528,9 +514,7 @@ async def _handle_context_batch( # noqa: C901 case AdjustmentMethod.FLAT: adjusted_expression_requirement = expression_requirement case _: - adjusted_expression_requirement = int( - custom_df.iloc[custom_df["context"] == context_name, "req"].iloc[0] - ) + adjusted_expression_requirement = int(custom_df.iloc[custom_df["context"] == context_name, "req"].iloc[0]) if adjusted_expression_requirement != expression_requirement: logger.debug( @@ -715,8 +699,7 @@ def _parse_args() -> _Arguments: required=False, default=None, dest="expression_requirement", - help="Number of sources with active gene for it to be considered active " - "even if it is not a high confidence-gene", + help="Number of sources with active gene for it to be considered active even if it is not a high confidence-gene", ) parser.add_argument( @@ -735,8 +718,7 @@ def _parse_args() -> _Arguments: required="custom" in sys.argv, # required if --requriement-adjust is "custom", dest="custom_expression_filename", default=None, - help="Name of .xlsx file where first column is context names and " - "second column is expression requirement for that context", + help="Name of .xlsx file where first column is context names and second column is expression requirement for that context", ) parser.add_argument( @@ -746,8 +728,7 @@ def _parse_args() -> _Arguments: required=False, default=False, dest="no_high_confidence", - help="Prevent high-confidence genes forcing a gene to be used in final model, " - "irrespective of other other data sources", + help="Prevent high-confidence genes forcing a gene to be used in final model, irrespective of other other data sources", ) parser.add_argument( diff --git a/main/como/proteomics/Crux.py b/main/como/proteomics/Crux.py index b9d0107f..9351cf4e 100644 --- a/main/como/proteomics/Crux.py +++ b/main/como/proteomics/Crux.py @@ -128,9 +128,7 @@ def mzml_to_sqt(self) -> None: ) # Replace all "comet.*" in output directory with the name of the file being processed - comet_files = [ - str(file) for file in os.listdir(file_information.sqt_base_path) if str(file).startswith("comet.") - ] + comet_files = [str(file) for file in os.listdir(file_information.sqt_base_path) if str(file).startswith("comet.")] for file_name in comet_files: # Determine the old file path old_file_path: Path = Path(file_information.sqt_base_path, file_name) @@ -242,9 +240,7 @@ def collect_uniprot_ids_and_ion_intensity(self) -> None: # Assign the file_information intensity dataframe to the gathered values self._file_information[i].intensity_df = pd.DataFrame(average_intensities_dict) - self._file_information[i].intensity_df = ( - self._file_information[i].intensity_df.groupby("uniprot", as_index=False).mean() - ) + self._file_information[i].intensity_df = self._file_information[i].intensity_df.groupby("uniprot", as_index=False).mean() async def _convert_uniprot_wrapper(self) -> None: """This function is a multiprocessing wrapper around the convert_ids function""" @@ -252,9 +248,7 @@ async def _convert_uniprot_wrapper(self) -> None: # Create a progress bar of results # From: https://stackoverflow.com/a/61041328/ - progress_bar = tqdm.tqdm( - desc="Starting UniProt to Gene Symbol conversion... ", total=len(self._file_information) - ) + progress_bar = tqdm.tqdm(desc="Starting UniProt to Gene Symbol conversion... ", total=len(self._file_information)) for i, result in enumerate(asyncio.as_completed(values)): await result # Get result from asyncio.as_completed progress_bar.set_description(f"Working on {i + 1} of {len(self._file_information)}") @@ -378,9 +372,7 @@ def split_abundance_values(self) -> None: # Create a new dataframe to split the S# columns from split_frame: pd.DataFrame = dataframe.copy() # Get the current S{i} columns in - abundance_columns: list[str] = [ - column for column in split_frame.columns if re.match(rf"{cell_type}_S{i}R\d+", column) - ] + abundance_columns: list[str] = [column for column in split_frame.columns if re.match(rf"{cell_type}_S{i}R\d+", column)] take_columns: list[str] = ["symbol"] + abundance_columns average_intensity_name: str = f"{cell_type}_S{i}" diff --git a/main/como/proteomics/FTPManager.py b/main/como/proteomics/FTPManager.py index 1be45d91..f846481d 100644 --- a/main/como/proteomics/FTPManager.py +++ b/main/como/proteomics/FTPManager.py @@ -18,9 +18,7 @@ from .FileInformation import FileInformation, clear_print -async def aioftp_client( - host: str, username: str = "anonymous", password: str = "guest", port: int = 21, max_attempts: int = 3 -) -> aioftp.Client: +async def aioftp_client(host: str, username: str = "anonymous", password: str = "guest", port: int = 21, max_attempts: int = 3) -> aioftp.Client: """This class is responsible for creating a "client" connection""" connection_successful: bool = False attempt_num: int = 1 diff --git a/main/como/proteomics/FileInformation.py b/main/como/proteomics/FileInformation.py index 0b37a4b6..59a4a2ee 100644 --- a/main/como/proteomics/FileInformation.py +++ b/main/como/proteomics/FileInformation.py @@ -68,9 +68,7 @@ def __init__( self.sqt_base_path: Path = sqt_path.parent if intensity_csv is None: - self.intensity_csv: Path = Path( - project.configs.data_dir, "data_matrices", cell_type, f"protein_abundance_matrix_{cell_type}.csv" - ) + self.intensity_csv: Path = Path(project.configs.data_dir, "data_matrices", cell_type, f"protein_abundance_matrix_{cell_type}.csv") else: self.intensity_csv: Path = intensity_csv diff --git a/main/como/proteomics/proteomics_preprocess.py b/main/como/proteomics/proteomics_preprocess.py index 85d78ec2..fe363655 100644 --- a/main/como/proteomics/proteomics_preprocess.py +++ b/main/como/proteomics/proteomics_preprocess.py @@ -159,9 +159,7 @@ def _gather_data(self): # Iterate through the URLs available for url, study in zip(ftp_urls, studies, strict=True): - ftp_data: FTPManager.Reader = FTPManager.Reader( - root_link=url, file_extensions=self._preferred_extensions - ) + ftp_data: FTPManager.Reader = FTPManager.Reader(root_link=url, file_extensions=self._preferred_extensions) urls = list(ftp_data.files) sizes = list(ftp_data.file_sizes) @@ -169,9 +167,7 @@ def _gather_data(self): # Iterate through all files and sizes found for url_## for file, size in zip(urls, sizes, strict=True): - self.file_information.append( - FileInformation(cell_type=cell_type, download_url=file, file_size=size, study=study) - ) + self.file_information.append(FileInformation(cell_type=cell_type, download_url=file, file_size=size, study=study)) def print_download_size(self): """Print the total size to download if we must download data.""" @@ -313,16 +309,11 @@ def parse_args() -> argparse.Namespace: if args.core_count == "all": args.core_count = os.cpu_count() elif not str(args.core_count).isdigit(): - raise ValueError( - f"Invalid option '{args.core_count}' for option '--cores'. Enter an integer or 'all' to use all cores" - ) + raise ValueError(f"Invalid option '{args.core_count}' for option '--cores'. Enter an integer or 'all' to use all cores") else: args.core_count = int(args.core_count) if args.core_count > os.cpu_count(): - logger.info( - f"{args.core_count} cores not available, system only has {os.cpu_count()} cores. " - f"Setting '--cores' to {os.cpu_count()}" - ) + logger.info(f"{args.core_count} cores not available, system only has {os.cpu_count()} cores. Setting '--cores' to {os.cpu_count()}") args.core_count = os.cpu_count() return args diff --git a/main/como/proteomics_gen.py b/main/como/proteomics_gen.py index d020d71d..c205aea4 100644 --- a/main/como/proteomics_gen.py +++ b/main/como/proteomics_gen.py @@ -137,10 +137,7 @@ def load_empty_dict(): return context_name, data else: - logger.warning( - f"Proteomics gene expression file for {context_name} was not found at {full_save_filepath}. " - f"Is this intentional?" - ) + logger.warning(f"Proteomics gene expression file for {context_name} was not found at {full_save_filepath}. Is this intentional?") return load_empty_dict() diff --git a/main/como/proteomics_preprocessing.py b/main/como/proteomics_preprocessing.py index bced01b0..4093dba0 100644 --- a/main/como/proteomics_preprocessing.py +++ b/main/como/proteomics_preprocessing.py @@ -36,15 +36,9 @@ def z_score_calc(abundance: pd.DataFrame, min_thresh: int) -> ZResult: # np.zeros((1000, len(abundance.columns)), dtype=np.float64), z_result = ZResult( - zfpkm=pd.DataFrame( - data=np.nan * np.ones_like(values), index=abundance.index, columns=abundance.columns, dtype=np.float64 - ), - x_range=pd.DataFrame( - data=np.zeros((1000, len(abundance.columns))), columns=abundance.columns, dtype=np.float64 - ), - density=pd.DataFrame( - data=np.zeros((1000, len(abundance.columns))), columns=abundance.columns, dtype=np.float64 - ), + zfpkm=pd.DataFrame(data=np.nan * np.ones_like(values), index=abundance.index, columns=abundance.columns, dtype=np.float64), + x_range=pd.DataFrame(data=np.zeros((1000, len(abundance.columns))), columns=abundance.columns, dtype=np.float64), + density=pd.DataFrame(data=np.zeros((1000, len(abundance.columns))), columns=abundance.columns, dtype=np.float64), mu=np.zeros(len(abundance.columns)), std_dev=np.zeros(len(abundance.columns)), max_fpkm_peak=np.zeros(len(abundance.columns)), @@ -98,9 +92,7 @@ def plot_gaussian_fit(z_results: ZResult, facet_titles: bool = True, x_min: int std_dev = z_results.std_dev max_fpkm = z_results.max_fpkm_peak - fig = make_subplots( - rows=len(zfpkm.columns), cols=1, subplot_titles=zfpkm.columns if facet_titles else [None] * len(zfpkm.columns) - ) + fig = make_subplots(rows=len(zfpkm.columns), cols=1, subplot_titles=zfpkm.columns if facet_titles else [None] * len(zfpkm.columns)) for i, col in enumerate(zfpkm.columns): fitted = stats.norm.pdf(x_range[col], loc=mu[i], scale=std_dev[i]) scale_fit = fitted * (max_fpkm[i] / fitted.max()) @@ -136,9 +128,7 @@ def protein_transform_main( output_z_score_matrix_filepath: Path, ) -> None: """Transform protein abundance data.""" - abundance_df: pd.DataFrame = ( - pd.read_csv(abundance_df) if isinstance(abundance_df, str | Path) else abundance_df.fillna(0) - ) + abundance_df: pd.DataFrame = pd.read_csv(abundance_df) if isinstance(abundance_df, str | Path) else abundance_df.fillna(0) abundance_df = abundance_df[np.isfinite(abundance_df).all(axis=1)] # Remove +/- infinity values z_transform: ZResult = z_score_calc(abundance_df, min_thresh=0) diff --git a/main/como/rnaseq_gen.py b/main/como/rnaseq_gen.py index b0079b8c..53da3c69 100644 --- a/main/como/rnaseq_gen.py +++ b/main/como/rnaseq_gen.py @@ -3,7 +3,8 @@ import math import multiprocessing import time -from collections import Callable, namedtuple +from collections import namedtuple +from collections.abc import Callable from dataclasses import dataclass, field from enum import Enum from functools import partial @@ -114,7 +115,7 @@ def k_over_a(k: int, a: float) -> Callable[[npt.NDArray], bool]: :param k: The minimum number of times the sum of elements must be greater than or equal to A. :param a: The value to compare the sum of elements to. :return: A function that accepts a NumPy array to perform the actual filtering - """ # noqa: E501 + """ def filter_func(row: npt.NDArray) -> bool: return np.sum(row >= a) >= k @@ -134,11 +135,7 @@ def genefilter(data: pd.DataFrame | npt.NDArray, filter_func: Callable[[npt.NDAr if not isinstance(data, (pd.DataFrame, npt.NDArray)): raise TypeError("Unsupported data type. Must be a Pandas DataFrame or a NumPy array.") - return ( - data.apply(filter_func, axis=1).values - if isinstance(data, pd.DataFrame) - else np.apply_along_axis(filter_func, axis=1, arr=data) - ) + return data.apply(filter_func, axis=1).values if isinstance(data, pd.DataFrame) else np.apply_along_axis(filter_func, axis=1, arr=data) async def _read_counts(path: Path) -> pd.DataFrame: @@ -236,7 +233,7 @@ def calculate_tpm(metrics: NamedMetrics) -> NamedMetrics: def calculate_fpkm(metrics: NamedMetrics) -> NamedMetrics: - """Calculate the Fragments Per Kilobase of transcript per Million mapped reads (FPKM) for each sample in the metrics dictionary.""" # noqa: E501 + """Calculate the Fragments Per Kilobase of transcript per Million mapped reads (FPKM) for each sample in the metrics dictionary.""" matrix_values = [] for study in metrics: for sample in range(metrics[study].num_samples): @@ -348,22 +345,18 @@ def zfpkm_transform( ) -> tuple[dict[str, _ZFPKMResult], DataFrame]: """Perform zFPKM calculation/transformation.""" if update_every_percent > 1: - logger.warning( - f"update_every_percent should be a decimal value between 0 and 1; got: {update_every_percent} - will convert to percentage" - ) + logger.warning(f"update_every_percent should be a decimal value between 0 and 1; got: {update_every_percent} - will convert to percentage") update_every_percent /= 100 total = len(fpkm_df.columns) update_per_step: int = int(np.ceil(total * update_every_percent)) cores = min(multiprocessing.cpu_count() - 2, total) logger.debug(f"Processing {total:,} samples through zFPKM transform using {cores} cores") - logger.debug( - f"Will update every {update_per_step:,} steps as this is approximately {update_every_percent:.1%} of {total:,}" - ) + logger.debug(f"Will update every {update_per_step:,} steps as this is approximately {update_every_percent:.1%} of {total:,}") with Pool(processes=cores) as pool: kernel = KernelDensity(kernel="gaussian", bandwidth=bandwidth) - chunksize = int(math.ceil(len(fpkm_df.columns) / (4 * cores))) + chunksize = math.ceil(len(fpkm_df.columns) / (4 * cores)) partial_func = partial(_zfpkm_calculation, kernel=kernel, peak_parameters=peak_parameters) chunk_time = time.time() start_time = time.time() @@ -452,9 +445,7 @@ def calculate_z_score(metrics: NamedMetrics) -> NamedMetrics: """Calculate the z-score for each sample in the metrics dictionary.""" for sample in metrics: log_matrix = np.log(metrics[sample].normalization_matrix) - z_matrix = pd.DataFrame( - data=sklearn.preprocessing.scale(log_matrix, axis=1), columns=metrics[sample].sample_names - ) + z_matrix = pd.DataFrame(data=sklearn.preprocessing.scale(log_matrix, axis=1), columns=metrics[sample].sample_names) metrics[sample].z_score_matrix = z_matrix return metrics @@ -498,11 +489,7 @@ def cpm_filter( top_samples = round(n_top * len(counts.columns)) # noqa: F841 test_bools = pd.DataFrame({"entrez_gene_ids": entrez_ids}) for i in range(len(counts_per_million.columns)): - cutoff = ( - 10e6 / (np.median(np.sum(counts[:, i]))) - if cut_off == "default" - else (1e6 * cut_off) / np.median(np.sum(counts[:, i])) - ) + cutoff = 10e6 / (np.median(np.sum(counts[:, i]))) if cut_off == "default" else (1e6 * cut_off) / np.median(np.sum(counts[:, i])) test_bools = test_bools.merge(counts_per_million[counts_per_million.iloc[:, i] > cutoff]) return metrics @@ -528,12 +515,8 @@ def tpm_quantile_filter(*, metrics: NamedMetrics, filtering_options: _FilteringO top_samples = round(n_top * len(tpm_matrix.columns)) tpm_quantile = tpm_matrix[tpm_matrix > 0] - quantile_cutoff = np.quantile( - a=tpm_quantile.values, q=1 - (cut_off / 100), axis=0 - ) # Compute quantile across columns - boolean_expression = pd.DataFrame( - data=tpm_matrix > quantile_cutoff, index=tpm_matrix.index, columns=tpm_matrix.columns - ).astype(int) + quantile_cutoff = np.quantile(a=tpm_quantile.values, q=1 - (cut_off / 100), axis=0) # Compute quantile across columns + boolean_expression = pd.DataFrame(data=tpm_matrix > quantile_cutoff, index=tpm_matrix.index, columns=tpm_matrix.columns).astype(int) min_func = k_over_a(min_samples, 0.9) top_func = k_over_a(top_samples, 0.9) @@ -548,9 +531,7 @@ def tpm_quantile_filter(*, metrics: NamedMetrics, filtering_options: _FilteringO metric.normalization_matrix = metrics[sample].normalization_matrix.iloc[min_genes, :] keep_top_genes = [gene for gene, keep in zip(entrez_ids, top_genes, strict=True) if keep] - metric.high_confidence_entrez_gene_ids = [ - gene for gene, keep in zip(entrez_ids, keep_top_genes, strict=True) if keep - ] + metric.high_confidence_entrez_gene_ids = [gene for gene, keep in zip(entrez_ids, keep_top_genes, strict=True) if keep] metrics = calculate_z_score(metrics) @@ -587,9 +568,7 @@ def zfpkm_filter(*, metrics: NamedMetrics, filtering_options: _FilteringOptions, top_samples = round(high_confidence_sample_expression * len(zfpkm_df.columns)) top_func = k_over_a(top_samples, cut_off) top_genes: npt.NDArray[bool] = genefilter(zfpkm_df, top_func) - metric.high_confidence_entrez_gene_ids = [ - gene for gene, keep in zip(metric.entrez_gene_ids, top_genes, strict=True) if keep - ] + metric.high_confidence_entrez_gene_ids = [gene for gene, keep in zip(metric.entrez_gene_ids, top_genes, strict=True) if keep] return metrics @@ -605,9 +584,7 @@ def filter_counts( """Filter the count matrix based on the specified technique.""" match technique: case FilteringTechnique.cpm: - return cpm_filter( - context_name=context_name, metrics=metrics, filtering_options=filtering_options, prep=prep - ) + return cpm_filter(context_name=context_name, metrics=metrics, filtering_options=filtering_options, prep=prep) case FilteringTechnique.tpm: return tpm_quantile_filter(metrics=metrics, filtering_options=filtering_options) case FilteringTechnique.zfpkm: @@ -666,9 +643,7 @@ async def _save_rnaseq_tests( top_genes.extend(metric.high_confidence_entrez_gene_ids) expression_frequency = pd.Series(expressed_genes).value_counts() - expression_df = pd.DataFrame( - {"entrez_gene_id": expression_frequency.index, "frequency": expression_frequency.values} - ) + expression_df = pd.DataFrame({"entrez_gene_id": expression_frequency.index, "frequency": expression_frequency.values}) expression_df["prop"] = expression_df["frequency"] / len(metrics) expression_df = expression_df[expression_df["prop"] >= filtering_options.batch_ratio] @@ -688,17 +663,13 @@ async def _save_rnaseq_tests( high_confidence_count = len(boolean_matrix[boolean_matrix["high"] == 1]) boolean_matrix.to_csv(output_filepath, index=False) - logger.info( - f"{context_name} - Found {expressed_count} expressed and {high_confidence_count} confidently expressed genes" - ) + logger.info(f"{context_name} - Found {expressed_count} expressed and {high_confidence_count} confidently expressed genes") logger.success(f"Wrote boolean matrix to {output_filepath}") async def _create_metadata_df(path: Path) -> pd.DataFrame: if path.suffix not in {".xls", ".xlsx"}: - raise ValueError( - f"Expected an excel file with extension of '.xlsx' or '.xls', got '{path.suffix}'. Attempted to process: {path}" - ) + raise ValueError(f"Expected an excel file with extension of '.xlsx' or '.xls', got '{path.suffix}'. Attempted to process: {path}") return pd.read_excel(path) @@ -746,9 +717,7 @@ async def rnaseq_gen( # noqa: C901, allow complex function if not input_metadata_df and not input_metadata_filepath: raise ValueError("At least one of input_metadata_filepath or input_metadata_df must be provided") - technique = ( - FilteringTechnique.from_string(str(technique.lower())) if isinstance(technique, (str, int)) else technique - ) + technique = FilteringTechnique.from_string(str(technique.lower())) if isinstance(technique, (str, int)) else technique match technique: case FilteringTechnique.tpm: diff --git a/main/como/rnaseq_preprocess.py b/main/como/rnaseq_preprocess.py index 4f53962c..dac056fe 100644 --- a/main/como/rnaseq_preprocess.py +++ b/main/como/rnaseq_preprocess.py @@ -131,7 +131,7 @@ def _organize_gene_counts_files(data_dir: Path) -> list[_StudyMetrics]: if len(gene_counts_directories) != len(strandedness_directories): raise ValueError( f"Unequal number of gene count directories and strandedness directories. " - f"Found {len(gene_counts_directories)} gene count directories and {len(strandedness_directories)} strandedness directories." # noqa: E501 + f"Found {len(gene_counts_directories)} gene count directories and {len(strandedness_directories)} strandedness directories." f"\nGene count directory: {gene_count_dir}\nStrandedness directory: {strand_dir}" ) @@ -172,11 +172,7 @@ async def _process_first_multirun_sample(strand_file: Path, all_counts_files: li run_counts = star_information.count_matrix[["ensembl_gene_id", strand_information]] run_counts.columns = pd.Index(["ensembl_gene_id", "counts"]) - sample_count = ( - run_counts - if sample_count.empty - else sample_count.merge(run_counts, on=["ensembl_gene_id", "counts"], how="outer") - ) + sample_count = run_counts if sample_count.empty else sample_count.merge(run_counts, on=["ensembl_gene_id", "counts"], how="outer") # Set na values to 0 sample_count = sample_count.fillna(value="0") @@ -261,9 +257,7 @@ async def _write_counts_matrix( ) -> pd.DataFrame: """Create a counts matrix file by reading gene counts table(s).""" study_metrics = _organize_gene_counts_files(data_dir=como_context_dir) - counts: list[pd.DataFrame] = await asyncio.gather( - *[_create_sample_counts_matrix(metric) for metric in study_metrics] - ) + counts: list[pd.DataFrame] = await asyncio.gather(*[_create_sample_counts_matrix(metric) for metric in study_metrics]) final_matrix = pd.DataFrame() for count in counts: @@ -357,9 +351,7 @@ async def _create_config_df( # noqa: C901 with layout_files[0].open("r") as file: layout = file.read().strip() elif len(layout_files) > 1: - raise ValueError( - f"Multiple matching layout files for {label}, make sure there is only one copy for each replicate in COMO_input" - ) + raise ValueError(f"Multiple matching layout files for {label}, make sure there is only one copy for each replicate in COMO_input") strand = "UNKNOWN" if len(strand_files) == 0: @@ -372,9 +364,7 @@ async def _create_config_df( # noqa: C901 with strand_files[0].open("r") as file: strand = file.read().strip() elif len(strand_files) > 1: - raise ValueError( - f"Multiple matching strandedness files for {label}, make sure there is only one copy for each replicate in COMO_input" - ) + raise ValueError(f"Multiple matching strandedness files for {label}, make sure there is only one copy for each replicate in COMO_input") prep = "total" if len(prep_files) == 0: @@ -385,9 +375,7 @@ async def _create_config_df( # noqa: C901 if prep not in ["total", "mrna"]: raise ValueError(f"Prep method must be either 'total' or 'mrna' for {label}") elif len(prep_files) > 1: - raise ValueError( - f"Multiple matching prep files for {label}, make sure there is only one copy for each replicate in COMO_input" - ) + raise ValueError(f"Multiple matching prep files for {label}, make sure there is only one copy for each replicate in COMO_input") mean_fragment_size = 100 if len(frag_files) == 0 and prep != RNAPrepMethod.TOTAL.value: @@ -415,9 +403,7 @@ async def _create_config_df( # noqa: C901 mean_fragment_size = sum(mean_fragment_sizes * library_sizes) / sum(library_sizes) elif len(frag_files) > 1: - raise ValueError( - f"Multiple matching fragment files for {label}, make sure there is only one copy for each replicate in COMO_input" - ) + raise ValueError(f"Multiple matching fragment files for {label}, make sure there is only one copy for each replicate in COMO_input") sample_names.append(f"{context_name}_{study_number}{rep_number}") fragment_lengths.append(mean_fragment_size) @@ -460,9 +446,7 @@ async def read_counts(file: Path) -> list[str]: ) return conversion["entrez_gene_id"].tolist() - logger.info( - "Fetching gene info (this may take 1-5 minutes depending on the number of genes and your internet connection)" - ) + logger.info("Fetching gene info (this may take 1-5 minutes depending on the number of genes and your internet connection)") genes = set(chain.from_iterable(await asyncio.gather(*[read_counts(f) for f in counts_matrix_filepaths]))) gene_data = await MyGene(cache=cache).query(items=list(genes), taxon=taxon, scopes="entrezgene") gene_info: pd.DataFrame = pd.DataFrame( @@ -486,13 +470,7 @@ async def read_counts(file: Path) -> list[str]: gene_info.at[i, "start_position"] = start_pos gene_info.at[i, "end_position"] = end_pos - gene_info = gene_info[ - ( - (gene_info["entrez_gene_id"] != "-") - & (gene_info["ensembl_gene_id"] != "-") - & (gene_info["gene_symbol"] != "-") - ) - ] + gene_info = gene_info[((gene_info["entrez_gene_id"] != "-") & (gene_info["ensembl_gene_id"] != "-") & (gene_info["gene_symbol"] != "-"))] gene_info["size"] = gene_info["end_position"].astype(int) - gene_info["start_position"].astype(int) gene_info.drop(columns=["start_position", "end_position"], inplace=True) gene_info.sort_values(by="ensembl_gene_id", inplace=True) @@ -556,11 +534,7 @@ async def _process( # create the gene info filepath based on provided data await _create_gene_info_file( - counts_matrix_filepaths=[ - f - for f in [*input_matrix_filepath, output_trna_matrix_filepath, output_mrna_matrix_filepath] - if f is not None - ], + counts_matrix_filepaths=[f for f in [*input_matrix_filepath, output_trna_matrix_filepath, output_mrna_matrix_filepath] if f is not None], output_filepath=output_gene_info_filepath, taxon=taxon, cache=cache, @@ -611,21 +585,13 @@ async def rnaseq_preprocess( output_gene_info_filepath = output_gene_info_filepath.resolve() como_context_dir = como_context_dir.resolve() input_matrix_filepath = [i.resolve() for i in _listify(input_matrix_filepath)] if input_matrix_filepath else None - output_trna_config_filepath = ( - output_trna_config_filepath.resolve() if output_trna_config_filepath else output_trna_config_filepath - ) - output_mrna_config_filepath = ( - output_mrna_config_filepath.resolve() if output_mrna_config_filepath else output_mrna_config_filepath - ) + output_trna_config_filepath = output_trna_config_filepath.resolve() if output_trna_config_filepath else output_trna_config_filepath + output_mrna_config_filepath = output_mrna_config_filepath.resolve() if output_mrna_config_filepath else output_mrna_config_filepath output_trna_count_matrix_filepath = ( - output_trna_count_matrix_filepath.resolve() - if output_trna_count_matrix_filepath - else output_trna_count_matrix_filepath + output_trna_count_matrix_filepath.resolve() if output_trna_count_matrix_filepath else output_trna_count_matrix_filepath ) output_mrna_count_matrix_filepath = ( - output_mrna_count_matrix_filepath.resolve() - if output_mrna_count_matrix_filepath - else output_mrna_count_matrix_filepath + output_mrna_count_matrix_filepath.resolve() if output_mrna_count_matrix_filepath else output_mrna_count_matrix_filepath ) input_matrix_filepath = _listify(input_matrix_filepath) diff --git a/main/como/utils.py b/main/como/utils.py index a359de0f..7e341d06 100644 --- a/main/como/utils.py +++ b/main/como/utils.py @@ -95,9 +95,7 @@ class Compartments: "s": ["eyespot", "eyespot apparatus", "stigma"], } - _REVERSE_LOOKUP: ClassVar[dict[str, list[str]]] = { - value.lower(): key for key, values in SHORTHAND.items() for value in values - } + _REVERSE_LOOKUP: ClassVar[dict[str, list[str]]] = {value.lower(): key for key, values in SHORTHAND.items() for value in values} @classmethod def get(cls, longhand: str) -> str | None: @@ -125,20 +123,14 @@ def stringlist_to_list(stringlist: str | list[str]) -> list[str]: new_list: list[str] = stringlist.strip("[]").replace("'", "").replace(" ", "").split(",") # Show a warning if more than one item is present in the list (this means we are using the old method) - logger.critical( - "DeprecationWarning: Please use the new method of providing context names, " - "i.e. --output-filetypes 'type1 type2 type3'." - ) + logger.critical("DeprecationWarning: Please use the new method of providing context names, i.e. --output-filetypes 'type1 type2 type3'.") logger.critical( "If you are using COMO, this can be done by setting the 'context_names' variable to a " "simple string separated by spaces. Here are a few examples!" ) logger.critical("context_names = 'cellType1 cellType2 cellType3'") logger.critical("output_filetypes = 'output1 output2 output3'") - logger.critical( - "\nYour current method of passing context names will be removed in the future. " - "Update your variables above accordingly!\n\n" - ) + logger.critical("\nYour current method of passing context names will be removed in the future. Update your variables above accordingly!\n\n") return new_list @@ -158,9 +150,7 @@ def split_gene_expression_data(expression_data: pd.DataFrame, recon_algorithm: A expression_data.loc[:, "entrez_gene_id"] = expression_data["entrez_gene_id"].astype(str) single_gene_names = expression_data[~expression_data["entrez_gene_id"].str.contains("//")] multiple_gene_names = expression_data[expression_data["entrez_gene_id"].str.contains("//")] - split_gene_names = multiple_gene_names.assign( - entrez_gene_id=multiple_gene_names["entrez_gene_id"].str.split("///") - ).explode("entrez_gene_id") + split_gene_names = multiple_gene_names.assign(entrez_gene_id=multiple_gene_names["entrez_gene_id"].str.split("///")).explode("entrez_gene_id") gene_expressions = pd.concat([single_gene_names, split_gene_names], axis=0, ignore_index=True) gene_expressions.set_index("entrez_gene_id", inplace=True) @@ -193,9 +183,7 @@ async def _format_determination( :return: A pandas DataFrame """ requested_output = [requested_output] if isinstance(requested_output, Output) else requested_output - cohersion = (await biodbnet.db_find(values=input_values, output_db=requested_output, taxon=taxon)).drop( - columns=["Input Type"] - ) + cohersion = (await biodbnet.db_find(values=input_values, output_db=requested_output, taxon=taxon)).drop(columns=["Input Type"]) cohersion.columns = pd.Index(["input_value", *[o.value.replace(" ", "_").lower() for o in requested_output]]) return cohersion diff --git a/ruff.toml b/ruff.toml index 3691008f..b65a8693 100644 --- a/ruff.toml +++ b/ruff.toml @@ -1,4 +1,4 @@ -line-length = 120 +line-length = 150 extend-include = ["docs/**/*.py", "tests/**/*.py", "**/*.ipynb"] exclude = ["__init__.py"] @@ -9,46 +9,46 @@ docstring-code-format = true [lint] # Linting rules: https://docs.astral.sh/ruff/rules/ select = [ - "A", # do not use python builtins for variables or parameters; https://pypi.org/project/flake8-builtins/ - "ASYNC", # identify asynchronous-related problems; https://pypi.org/project/flake8-async/ - "B", # find likely bugs and design problems; https://pypi.org/project/flake8-bugbear/ - "C4", # create better list, set, & dict comprehensions; https://pypi.org/project/flake8-comprehensions/ - "C90", # check for complexity; https://pypi.org/project/mccabe/ - "D", # docstring style checker; https://pypi.org/project/pydocstyle/ - "DOC", # docstring linter; https://pypi.org/project/pydoclint/ - "E", # style guide checking; https://pypi.org/project/pycodestyle/ - "F", # check for errors; https://pypi.org/project/pyflakes/ - "RUF", # ruff-specific linting rules; https://docs.astral.sh/ruff/rules/#ruff-specific-rules-ruf - "FA", # use from __future__ import annotations if needed; https://pypi.org/project/flake8-future-annotations/ - "FURB", # refurbish and modernize Python codebases; https://pypi.org/project/refurb/ - "I", # sorting imports rules; https://pypi.org/project/isort/ - "N", # check naming conventions; https://pypi.org/project/pep8-naming/ - "PERF", # check performance anti-patterns; https://pypi.org/project/perflint/ - "PT", # check common style issues or inconsistencies in pytest; https://pypi.org/project/flake8-pytest-style/ - "PTH", # use pathlib where possible; https://pypi.org/project/flake8-use-pathlib/ - "S", # security testing; https://pypi.org/project/flake8-bandit/ - "SIM", # check for code that can be simplified; https://pypi.org/project/flake8_simplify/ - "T20", # do not use prints in production; https://pypi.org/project/flake8-print/ - "TRY", # prevent Exception handling anti-patterns; https://pypi.org/project/tryceratops/ + "A", # do not use python builtins for variables or parameters; https://pypi.org/project/flake8-builtins/ + "ASYNC", # identify asynchronous-related problems; https://pypi.org/project/flake8-async/ + "B", # find likely bugs and design problems; https://pypi.org/project/flake8-bugbear/ + "C4", # create better list, set, & dict comprehensions; https://pypi.org/project/flake8-comprehensions/ + "C90", # check for complexity; https://pypi.org/project/mccabe/ + "D", # docstring style checker; https://pypi.org/project/pydocstyle/ + "DOC", # docstring linter; https://pypi.org/project/pydoclint/ + "E", # style guide checking; https://pypi.org/project/pycodestyle/ + "F", # check for errors; https://pypi.org/project/pyflakes/ + "RUF", # ruff-specific linting rules; https://docs.astral.sh/ruff/rules/#ruff-specific-rules-ruf + "FA", # use from __future__ import annotations if needed; https://pypi.org/project/flake8-future-annotations/ + "FURB", # refurbish and modernize Python codebases; https://pypi.org/project/refurb/ + "I", # sorting imports rules; https://pypi.org/project/isort/ + "N", # check naming conventions; https://pypi.org/project/pep8-naming/ + "PERF", # check performance anti-patterns; https://pypi.org/project/perflint/ + "PT", # check common style issues or inconsistencies in pytest; https://pypi.org/project/flake8-pytest-style/ + "PTH", # use pathlib where possible; https://pypi.org/project/flake8-use-pathlib/ + "S", # security testing; https://pypi.org/project/flake8-bandit/ + "SIM", # check for code that can be simplified; https://pypi.org/project/flake8_simplify/ + "T20", # do not use prints in production; https://pypi.org/project/flake8-print/ + "TRY", # prevent Exception handling anti-patterns; https://pypi.org/project/tryceratops/ "UP" # upgrade syntax for newer versions; https://pypi.org/project/pyupgrade ] ignore = [ - "D100", # allow undocumented public module definitions - "D101", # allow undocumented public class - "D203", # do not require one blank line before class docstring - "D213", # first docstring line should be on the second line - "TRY003", # allow exception messages outside the `Exception` class - "F401", # allow unused imports + "D100", # allow undocumented public module definitions + "D101", # allow undocumented public class + "D203", # do not require one blank line before class docstring + "D213", # first docstring line should be on the second line + "TRY003", # allow exception messages outside the `Exception` class + "F401", # allow unused imports ] [lint.per-file-ignores] "tests/*" = [ - "D101", # allow undocumented public class - "D102", # allow undocumented class method - "D103", # allow undocumented public method definitions - "F811", # allow redefinition of variables, required for pytest fixtures - "S101", # allow use of `assert` in test files + "D101", # allow undocumented public class + "D102", # allow undocumented class method + "D103", # allow undocumented public method definitions + "F811", # allow redefinition of variables, required for pytest fixtures + "S101", # allow use of `assert` in test files ] "main/COMO.ipynb" = [ - "E501", # allow long lines + "E501", # allow long lines ] diff --git a/tests/unit/test_rnaseq_preprocess.py b/tests/unit/test_rnaseq_preprocess.py index c0d49c2e..56674d43 100644 --- a/tests/unit/test_rnaseq_preprocess.py +++ b/tests/unit/test_rnaseq_preprocess.py @@ -40,7 +40,7 @@ async def test_build_from_tab_valid_file(self): @pytest.mark.asyncio async def test_build_from_tab_invalid_file(self): """Validate error on invalid file.""" - with pytest.raises(ValueError, match="Building STAR information requires a '.tab' file"): + with pytest.raises(ValueError, match=r"Building STAR information requires a '.tab' file"): await _STARinformation.build_from_tab(TestSTARInformation.invalid_data)