diff --git a/docs/assets/data/breadcrumbs.json b/docs/assets/data/breadcrumbs.json index e5c2500e..aa4b42a4 100644 --- a/docs/assets/data/breadcrumbs.json +++ b/docs/assets/data/breadcrumbs.json @@ -165,6 +165,55 @@ "Bell User Guide", "Frequently Asked Questions" ], + "/userguides/negishi/": [ + "Home", + "Negishi User Guide" + ], + "/userguides/negishi/overview/": [ + "Home", + "Negishi User Guide", + "Negishi Overview" + ], + "/userguides/negishi/biography/": [ + "Home", + "Negishi User Guide", + "Biography of Negishi" + ], + "/userguides/negishi/accounts/": [ + "Home", + "Negishi User Guide", + "Accounts" + ], + "/userguides/negishi/software/": [ + "Home", + "Negishi User Guide", + "Software" + ], + "/userguides/negishi/run_jobs/": [ + "Home", + "Negishi User Guide", + "Running Jobs" + ], + "/userguides/negishi/storage/": [ + "Home", + "Negishi User Guide", + "File Storage and Transfer" + ], + "/userguides/negishi/gateway/": [ + "Home", + "Negishi User Guide", + "Gateway (Open OnDemand)" + ], + "/userguides/negishi/compile/": [ + "Home", + "Negishi User Guide", + "Compiling Source Code" + ], + "/userguides/negishii/faqs/": [ + "Home", + "Negishi User Guide", + "Frequently Asked Questions" + ], "/userguides/gautschi/": [ "Home", "Gautschi User Guide" @@ -972,4 +1021,4 @@ "Home", "FAQs" ] -} \ No newline at end of file +} diff --git a/docs/assets/images/userguides/negishi/bio_negishi.jpg b/docs/assets/images/userguides/negishi/bio_negishi.jpg new file mode 100644 index 00000000..94383998 Binary files /dev/null and b/docs/assets/images/userguides/negishi/bio_negishi.jpg differ diff --git a/docs/userguides/anvil/access.md b/docs/userguides/anvil/access.md index b0b8f28f..61d07d2c 100644 --- a/docs/userguides/anvil/access.md +++ b/docs/userguides/anvil/access.md @@ -2,8 +2,10 @@ tags: - Anvil - ACCESS + - NAIRR authors: - jin456 + - hkashgar search: boost: 2 draft: true @@ -13,13 +15,15 @@ draft: true ## Obtaining an Account -Anvil is an ACCESS computing resource. To use Anvil, you must first create an ACCESS account and request an allocation through the [ACCESS Allocation Request System](https://allocations.access-ci.org/). +Anvil is an ACCESS an NAIRR computing resource. To use Anvil, you must first create an ACCESS account and request an allocation through the [ACCESS Allocation Request System](https://allocations.access-ci.org/) or [NAIRR Pilot](https://nairrpilot.org/). !!! tip "New users to existing projects" - If you are joining an existing project, all you need to do is [login or create your ACCESS account](https://allocations.access-ci.org/) and the manager of your project can add you themselves! + If you are joining an existing project with an **[ACCESS allocation](#access)**, [login or create your ACCESS account](https://allocations.access-ci.org/) and the project manager can add you through the ACCESS portal. If your project has a **[NAIRR allocation](#nairr)**, you still need an [ACCESS account](https://allocations.access-ci.org/), but user management and allocation administration are handled by the project manager through the [NAIRR Pilot portal](https://submit-nairr.xras.org/).
+## ACCESS + ### What is ACCESS? [Advanced Cyberinfrastructure Coordination Ecosystem: Services & Support (ACCESS)](https://access-ci.org) is an NSF-funded program that manages access to the national research cyberinfrastructure (CI) resources. Any researcher who seeks to use one of these CI resources must follow ACCESS processes to get onto these resources. @@ -28,11 +32,9 @@ Anvil is an ACCESS computing resource. To use Anvil, you must first create an AC ACCESS coordinates a diverse set of resources including Anvil and other traditional HPC resources suited for resource-intensive CPU workloads, modern accelerator-based systems (e.g., GPU), as well as cloud resources. **Anvil provides both CPU and GPU resources as part of ACCESS.** A comprehensive list of all the ACCESS-managed resources can be found here along with their descriptions and ideal workloads: [https://allocations.access-ci.org/resources](https://allocations.access-ci.org/resources). -
- -## Access to Anvil +### Access to Anvil through ACCESS program -### Applying for a New Project +#### Applying for a New Project 1. Sign up for an ACCESS account (if you don’t have one already) at [https://allocations.access-ci.org](https://allocations.access-ci.org). 2. Prepare an allocation request with details of your proposed computational workflows (science, software needs), resource requirements, and a short CV. See the individual “Preparing Your … Request” pages for details on what documents are required: [https://allocations.access-ci.org/prepare-requests](https://allocations.access-ci.org/prepare-requests). @@ -41,7 +43,7 @@ ACCESS coordinates a diverse set of resources including Anvil and other traditio !!! tip "Help with proposal" Interested parties may contact the [ACCESS support](https://support.access-ci.org/) for help with an Anvil proposal. -### Joining an Existing Project +#### Joining an Existing Project 1. Sign up for an ACCESS account (if you don’t have one already) at [https://allocations.access-ci.org](https://allocations.access-ci.org). 2. Have the manager of the project add you through their admin portal. That simple! @@ -55,7 +57,7 @@ ACCESS coordinates a diverse set of resources including Anvil and other traditio You will use your Anvil username when logging into Anvil systems via SSH or your ACCESS username when using Open OnDemand. -### Which ACCESS tier should I choose? +#### Which ACCESS tier should I choose? As you can gather from [https://allocations.access-ci.org/project-types](https://allocations.access-ci.org/project-types), there are **four different tiers in ACCESS**. Broadly, these tiers provide increasing computational resources with corresponding stringent documentation and resource justification requirements. Furthermore, while Explore and Discover tier requests are reviewed on a rolling basis as they are submitted, Accelerate requests will be reviewed monthly and Maximize will be reviewed twice a year. The review period reflects the level of resources provided, and Explore and Discover applications are generally reviewed within a week. An important point to note is that ACCESS does not award you time on a specific computational resource (except for the Maximize tier). Users are awarded a certain number of ACCESS credits which they then exchange for time on a particular resource. Here are some guidelines on how to choose between the tiers: @@ -64,7 +66,7 @@ As you can gather from [https://allocations.access-ci.org/project-types](https:/ 3. If you would like to run simulations across multiple resources to identify the one best suited for you, Discover will provide you with sufficient credits to exchange across multiple systems. 4. One way of determining the appropriate tier is to determine what the credits would translate to in terms of computational resources. The [exchange calculator](https://allocations.access-ci.org/exchange_calculator) can be used to calculate what a certain number of ACCESS credits translates to in terms of “core-hours” or “GPU-hours” or “node-hours” on an ACCESS resource. For example: the maximum 400,000 ACCESS credits that you may be awarded in the Explore tier translates to ~334,000 CPU core hours or ~6000 GPU hours on Anvil. Based on the scale of simulations you would like to run, you may need to choose one tier or the other. -### What else should I know? +#### What else should I know? 1. You may request a **separate** allocation for each of your research grants and the allocation can last the duration of the grant (except for the Maximize tier which only lasts for 12 months). Allocations that do not cite a grant will last for 12 months. 2. **Supplements are not allowed** for Explore, Discover, and Accelerate tiers. Instead, you will need to move to a different tier if you require more resources. @@ -74,6 +76,47 @@ As you can gather from [https://allocations.access-ci.org/project-types](https:/ 6. You will also need to go to the allocations page and add any users you would like to have access to these resources. Note that they will need to sign up for ACCESS accounts as well before you can add them. 7. For other questions you may have, take a look at [ACCESS policies](https://allocations.access-ci.org/allocations-policy). + +
+ +## NAIRR +### What is NAIRR? + +The [National Artificial Intelligence Research Resource (NAIRR)](https://nairrpilot.org/) is a shared national research infrastructure that connects U.S. researchers and educators to AI resources to advance research, discovery, and innovation. The NAIRR Pilot is led by the U.S. National Science Foundation in collaboration with federal agency partners and non-governmental organizations. + + +### Which Anvil resources are available via NAIRR? + +NAIRR users can access Anvil's CPU, GPU, and AI resources for AI/ML research: + +- **Anvil GPU**: Nodes with AMD EPYC™ 7763 CPUs and 4 NVIDIA A100 GPUs (40GB each) +- **Anvil AI**: Nodes with Intel Xeon Platinum 8468 CPUs and 4 NVIDIA H100 GPUs (80GB each) + +
+ + +### Access to Anvil through NAIRR program + +Anvil is also available as a resource through the [National AI Research Resource (NAIRR) Pilot](https://nairrpilot.org/). **Anvil GPU** (NVIDIA A100) and **Anvil AI** (NVIDIA H100) nodes are accessible through the NAIRR Pilot for AI research and research that employs AI. + +#### Applying for NAIRR Access + +To request access to Anvil through the NAIRR Pilot: + +1. Sign up for an ACCESS account (if you don’t have one already) at [https://allocations.access-ci.org](https://allocations.access-ci.org). +2. Visit the [NAIRR Pilot portal](https://submit-nairr.xras.org/login). +2. Create a new proposal request for start up there is 1 page required proposal and for research resource there is 3 page one. +3. Submit an allocation request under the appropriate opportunity (such as Research Resources or Start-Up Projects). +4. Select Anvil GPU or Anvil AI as the target resource for your AI research also justify in you proposal on how you will use the resource and why you need the specific GPU or CPU hours. + +For more details on available opportunities and application procedures, see [https://nairrpilot.org/](https://nairrpilot.org/). + +!!! note "NAIRR Acknowledgment" + When you publish research based on NAIRR Pilot-provided resources, please include the following acknowledgment: + + *"This research is supported by the National Artificial Intelligence Research Resource (NAIRR) Pilot and the Anvil supercomputer supported by the National Science Foundation (award NSF-OAC 2005632) at Purdue University."* + + ## Helpful Tips We will strive to ensure that Anvil serves as a valuable resource to the national research community. We hope that you the user will assist us by making note of the following: diff --git a/docs/userguides/anvil/getting-started.md b/docs/userguides/anvil/getting-started.md index 35bd6733..cfad02e8 100644 --- a/docs/userguides/anvil/getting-started.md +++ b/docs/userguides/anvil/getting-started.md @@ -11,7 +11,9 @@ search: ## Start here -Before you can log in to Anvil, ensure you have: +Before you can log in to Anvil, ensure you have completed the steps for your allocation type: + +**For ACCESS allocations:** 1. Created your [ACCESS account](https://access-ci.org) 2. Applied for a project @@ -19,7 +21,14 @@ Before you can log in to Anvil, ensure you have: 4. [Transferred your credits to Anvil](https://allocations.access-ci.org/how-to#manage-an-explore-discover-or-accelerate-project) !!! warning "Anvil account creation" - Unless you have **transferred** credits onto Anvil, you will not have a valid Anvil username. This will result in errors that may look like: `failed to map user `. If you are not the PI of the project, ensure the PI has added you to the project and exchanged credits to Anvil. + For ACCESS allocations, unless you have **transferred** credits onto Anvil, you will not have a valid Anvil username. This will result in errors that may look like: `failed to map user `. If you are not the PI of the project, ensure the PI has added you to the project and exchanged credits to Anvil. For NAIRR allocations, your account is created automatically when your allocation is approved. + +**For NAIRR allocations:** + +1. Create your [ACCESS account](https://access-ci.org) +2. Apply for a project through the [NAIRR Pilot](https://nairrpilot.org/) +3. Your allocation on Anvil is approved (no credit transfer required with NAIRR—when your NAIRR proposal is accepted for Anvil resources, or when your transfer request to Anvil is accepted, your allocation is automatically available) +4. For more information on NAIRR allocation management, see the [NAIRR section](access.md#nairr). ## Logging in diff --git a/docs/userguides/anvil/overview.md b/docs/userguides/anvil/overview.md index 9ea0d34c..aac21980 100644 --- a/docs/userguides/anvil/overview.md +++ b/docs/userguides/anvil/overview.md @@ -3,6 +3,7 @@ tags: - Anvil authors: - jin456 + - hkashgar search: boost: 2 draft: false @@ -18,7 +19,7 @@ Anvil, which is funded by a $10 million award from the National Science Foundati The name "Anvil" reflects the Purdue Boilermakers' strength and workmanlike focus on producing results, and the Anvil supercomputer enables important discoveries across many different areas of science and engineering. Anvil also serves as an experiential learning laboratory for students to gain real-world experience using computing for their science, and for student interns to work with the Anvil team for construction and operation. We are training the research computing practitioners of the future. -Anvil is built in partnership with Dell and AMD and consists of 1,000 nodes with two 64-core AMD Epyc "Milan" processors each and delivers over 1 billion CPU core hours to ACCESS each year, with a peak performance of 5.3 petaflops. Anvil's nodes are interconnected with **100 Gbps Mellanox HDR InfiniBand**. The supercomputer ecosystem also includes 32 large memory nodes, each with 1 TB of RAM, and 16 nodes each with four **NVIDIA A100 Tensor Core GPUs** providing 1.5 PF of single-precision performance to support machine learning and artificial intelligence applications. +Anvil is built in partnership with Dell and AMD and consists of 1,000 nodes with two 64-core AMD Epyc "Milan" processors each and delivers over 1 billion CPU core hours to ACCESS each year, with a peak performance of 5.3 petaflops. Anvil's nodes are interconnected with **100 Gbps Mellanox HDR InfiniBand**. The supercomputer ecosystem also includes 32 large memory nodes, each with 1 TB of RAM, 16 nodes each with four **NVIDIA A100 Tensor Core GPUs**, and 21 nodes each with four **NVIDIA H100 Tensor Core GPUs** providing accelerated performance to support machine learning and artificial intelligence applications.
![anvil_glance](../../assets/images/userguides/anvil/anvil-glance.png){ width="800" } @@ -41,6 +42,7 @@ All Anvil nodes have 128 processor cores, 256 GB to 1 TB of RAM, and 100 Gbps In |A |1,000 |Two Milan CPUs @ 2.45GHz|128 |256GB | |B |32 |Two Milan CPUs @ 2.45GHz|128 |1TB | |G |16 |Two Milan CPUs @ 2.45GHz + Four NVIDIA A100 GPUs|128 |512GB | +|H |21 |Dual Intel Xeon Platinum 8468 CPUs + Four NVIDIA H100 GPUs|96 |1TB | Anvil nodes run CentOS 8 (Rocky Linux) and use Slurm (Simple Linux Utility for Resource Management) as the batch scheduler for resource and job management. The application of operating system patches will occur as security needs dictate. All nodes allow for unlimited stack usage, as well as unlimited core dump size (though disk space and server quotas may still be a limiting factor). diff --git a/docs/userguides/negishi/accounts.md b/docs/userguides/negishi/accounts.md new file mode 100644 index 00000000..235dd26a --- /dev/null +++ b/docs/userguides/negishi/accounts.md @@ -0,0 +1,27 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: negishi +search: + boost: 2 +--- + + +{% set resource = "negishi" %} + +{{ accounts_md_snippet(resource) }} + +## SSH Keys +{{ ssh_keys_snippet(resource) }} + +## SSH X11 Forwarding +{{ ssh_x11_snippet(resource) }} + +## Thinlinc +{{ thinlinc_snippet(resource) }} + +## Purchasing Nodes + +--8<-- "docs/snippets/purchase_nodes.md" diff --git a/docs/userguides/negishi/biography.md b/docs/userguides/negishi/biography.md new file mode 100644 index 00000000..db49fe23 --- /dev/null +++ b/docs/userguides/negishi/biography.md @@ -0,0 +1,17 @@ +--- +tags: + - Negishi +authors: + - hkashgar +search: + boost: 2 +--- + +# Biography of Ei-ichi Negishi + +![Portrait of Ei-ichi Negishi](../../assets/images/userguides/negishi/bio_negishi.jpg){ align=right } + + +Ei-ichi Negishi (1935-2021) was the Herbert C. Brown Distinguished Professor in the Department of Chemistry at Purdue. He came to Purdue in 1966 as a postdoctoral researcher in the lab of the Late Herbert C. Brown, and published 33 papers with Prof. Brown up through the time that Prof. Brown was awarded the Nobel Prize in Chemistry in 1979. With the award of the Nobel to Ei-ichi Negishi in 2010, Purdue has the rare distinction of a pair of Nobel Prize awards in two closely related areas. Professor Negishi’s Nobel Prize was awarded in recognition of his work on palladium-catalyzed cross-coupling chemistry (known world-– wide as the Negishi coupling). That work was described by the Nobel Foundation as "great art in a test tube". This is certainly appropriate as great scientists regard themselves as artists and explorers. The impact of that work was widespread, as it had been used in synthetic organic chemistry research worldwide, as well as in the commercial production of an array of pharmaceuticals and molecules used in the electronics industry. In recognition of and consistent with this idea, Ei-ichi and co-recipient Akira Suzuki were recently awarded Japan's highest cultural award, the "Order of Culture", bestowed in Nov. 2010 by the Emperor. + +Professor Negishi was a prolific researcher, with ~400 publications on an array of problems in synthetic organic chemistry, leading to numerous awards. To name just a few, the list includes the Chemical Society of Japan Award (1997), the American Chemical Society Award in Organometallic Chemistry (1998), the McCoy Award (1998), the Sigma Xi Award at Purdue (2003), the Nobel Prize in Chemistry (2010), the Order of Culture in Japan (2010), the American Chemical Society Award for Creative Work in Synthetic Organic Chemistry (2010), the Indiana Sagamore of the Wabash (2011) and the Purdue Order of the Griffin (2011). He was elected to the American Academy of Arts and Sciences in 2011. Professor Negishi was leading the Negishi-Brown Institute, which had continued his work on catalytic organic synthesis. Dr. Negishi was passionate about the prospects for catalytic approaches to the reduction of carbon dioxide to enable large scale production of useful products from this environmental waste product. It is very fitting that Purdue bestow an honorary doctorate degree on Professor Negishi, whose accomplishments and contributions will have a permanent impact on Purdue’s stature and global recognition. diff --git a/docs/userguides/negishi/compile.md b/docs/userguides/negishi/compile.md new file mode 100644 index 00000000..71a27f9b --- /dev/null +++ b/docs/userguides/negishi/compile.md @@ -0,0 +1,23 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Compiling Source Code + +Documentation on compiling source code on Negishi. + +## In This Section + +- [Compiling Serial Programs](compile/serial.md) +- [Compiling MPI Programs](compile/mpi.md) +- [Compiling OpenMP Programs](compile/openmp.md) +- [Compiling Hybrid Programs](compile/hybrid.md) +- [Intel MKL Library](compile/intel_mkl.md) + +[**Back to Negishi User Guide**](index.md) diff --git a/docs/userguides/negishi/compile/hybrid.md b/docs/userguides/negishi/compile/hybrid.md new file mode 100644 index 00000000..aaff08cc --- /dev/null +++ b/docs/userguides/negishi/compile/hybrid.md @@ -0,0 +1,13 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/compile_hybrid.md" + +[**Back to the Compiling Source Code section**](../compile.md) diff --git a/docs/userguides/negishi/compile/intel_mkl.md b/docs/userguides/negishi/compile/intel_mkl.md new file mode 100644 index 00000000..596359c9 --- /dev/null +++ b/docs/userguides/negishi/compile/intel_mkl.md @@ -0,0 +1,13 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/compile_intel_mkl.md" + +[**Back to the Compiling Source Code section**](../compile.md) diff --git a/docs/userguides/negishi/compile/mpi.md b/docs/userguides/negishi/compile/mpi.md new file mode 100644 index 00000000..80d8b839 --- /dev/null +++ b/docs/userguides/negishi/compile/mpi.md @@ -0,0 +1,13 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/compile_mpi.md" + +[**Back to the Compiling Source Code section**](../compile.md) diff --git a/docs/userguides/negishi/compile/openmp.md b/docs/userguides/negishi/compile/openmp.md new file mode 100644 index 00000000..45930916 --- /dev/null +++ b/docs/userguides/negishi/compile/openmp.md @@ -0,0 +1,13 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/compile_openmp.md" + +[**Back to the Compiling Source Code section**](../compile.md) diff --git a/docs/userguides/negishi/compile/serial.md b/docs/userguides/negishi/compile/serial.md new file mode 100644 index 00000000..4164621b --- /dev/null +++ b/docs/userguides/negishi/compile/serial.md @@ -0,0 +1,13 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/compile_serial.md" + +[**Back to the Compiling Source Code section**](../compile.md) diff --git a/docs/userguides/negishi/faqs.md b/docs/userguides/negishi/faqs.md new file mode 100644 index 00000000..8df469e1 --- /dev/null +++ b/docs/userguides/negishi/faqs.md @@ -0,0 +1,450 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Frequently Asked Questions + +Some common questions, errors, and problems are categorized below. You can also use the search box above to search the user guide for any issues you are seeing. + +## About Negishi + +### Can you remove me from the Negishi mailing list? + +Your subscription in the Negishi mailing list is tied to your account on Negishi. If you are no longer using your account on Negishi, your account can be deleted from the [My Accounts](https://www.rcac.purdue.edu/account/myinfo) page. Hover over the resource you wish to remove yourself from and click the red 'X' button. Your account and mailing list subscription will be removed overnight. Be sure to make a copy of any data you wish to keep first. + +### How is Negishi different than other Community Clusters? + +Negishi differs from the previous Community Clusters in several significant aspects: + +* Host naming convention in the Negishi cluster is different from earlier Community Clusters. Everything Negishi-related is contained within a `a003.negishi.rcac.purdue.edu` subdomain. Front-end login nodes are now named `loginNN` (as opposed to earlier `-feNN`), and compute nodes of each type `X` are named `xNNN` (as opposed to `-xNNN`). +* Negishi OnDemand Gateway is at the [gateway.negishi.rcac.purdue.edu](https://gateway.negishi.rcac.purdue.edu) (as opposed to earlier `gateway.negishi.rcac.purdue.edu` convention). +* Negishi home directories are entirely separate from other Community Clusters home directories. There is no automatic copying or synchronization between the two. At their discretion, users can copy parts or all of the Community Clusters home directory into Negishi - [instructions are provided](../storage/copyhome). +* Negishi contains the 3rd generation of AMD EPYC processors, codenamed "Milan". These CPUs support AVX2 vector instructions set. When compiling your code, use of `-march=znver3` flag (for latest GCC, Clang and AOCC compilers) or `-march=core-avx2` (for Intel compilers and GCC prior to 11.0) is recommended. +* GCC compiler with OpenMPI or MVAPICH2 MPI libraries are recommended for software development on Negishi. You can enable this software with `module load gcc openmpi` (default) or `module load gcc mvapich2`. +* If you use Jupyter notebooks, JupyterHub on Negishi will be available only via the [OnDemand Gateway](https://gateway.negishi.rcac.purdue.edu) rather than the freestanding version as on some previous systems. Other RCAC systems will transition to OnDemand as well, following Negishi. + +Upcoming 2023 +------------- + +* A subset of Negishi compute nodes contain ${resource.gpuname} accelerator cards which can significantly improve performance of compute-intensive workloads. These can be utilized by submitting jobs to the `${resource.gpuqueue}` queue (add `-A ${resource.gpuqueue}` to your job submission command). +* A selection of GPU-enabled ROCm application containers from the AMD InfinityHub collection is installed. + + +### Do I need to do anything to my firewall to access Negishi? + +No firewall changes are needed to access Negishi. However, to access data through Network Drives (i.e., CIFS, "Z: Drive"), you must be on a Purdue campus network or connected through [VPN](http://www.itap.purdue.edu/connections/vpn/). + +### Does Negishi have the same home directory as other clusters? + +The Negishi home directory and its contents are exclusive to Negishi cluster front-end hosts and compute nodes. This home directory is not available on other RCAC machines but Negishi. There is no automatic copying or synchronization between home directories. + +At your discretion you can manually copy all or parts of your main research computing home to Negishi using one of the [suggested methods](storage.md#file-transfer). + +If you plan to use `hsi` or `htar` commands to access Fortress tape archive from Negishi, please see also the [keytab generation question](#hsihtar-unable-to-authenticate-user-with-remote-gateway-error-2-or-9) for a temporary workaround to a potential caveat, while a permanent mitigation is being developed. + +## Logging In & Accounts + +### /usr/bin/xauth: error in locking authority file + +#### Problem + +I receive this message when logging in: + +`/usr/bin/xauth: error in locking authority file` + +#### Solution + +Your home directory disk quota is full. You may check your quota with `myquota`. + +You will need to free up space in your home directory. + +`ncdu` command is a convenient interactive tool to examine disk usage. Consider running `ncdu $HOME` to analyze where the bulk of the usage is. With this knowledge, you could then archive your data elsewhere (e.g. your research group's Data Depot space, or Fortress tape archive), or delete files you no longer need. + +There are several common locations that tend to grow large over time and are merely cached downloads. The following are safe to delete if you see them in the output of `ncdu $HOME`: + +``` +/home/myusername/.local/share/Trash +/home/myusername/.cache/pip +/home/myusername/.conda/pkgs +/home/myusername/.apptainer/cache +``` + +### My SSH connection hangs + +#### Problem + +Your console hangs while trying to connect to a RCAC Server. + +#### Solution + +This can happen due to various reasons. Most common reasons for hanging SSH terminals are: + +* **Network:** If you are connected over wifi, make sure that your Internet connection is fine. +* **Busy front-end server:** When you connect to a cluster, you SSH to one of the front-end login nodes. Due to transient user loads, one or more of the front-ends may become unresponsive for a short while. To avoid this, try reconnecting to the cluster or wait until the login node you have connected to has reduced load. +* **File system issue:** If a server has issues with one or more of the file systems (`home`, `scratch`, or `depot`) it may freeze your terminal. To avoid this you can connect to another front-end. + +If neither of the suggestions above work, please [contact support](https://www.rcac.purdue.edu/help) specifying the name of the server where your console is hung. + +### ThinLinc session frozen + +#### Problem + +Your ThinLinc session is frozen and you can not launch any commands or close the session. + +#### Solution + +This can happen due to various reasons. The most common reason is that you ran something memory-intensive inside that ThinLinc session on a front-end, so parts of the ThinLinc session got killed by Cgroups, and the entire session got stuck. + +* **If you are using a web-version ThinLinc remote desktop (inside the browser):** + + The web version does not have the capability to kill the existing session, only the standalone client does. Please install the standalone client and follow the steps below: + + [ThinLinc](accounts.md#thinlinc) + +* **If you are using a ThinLinc client:** + + Close the ThinLinc client, reopen the client login popup, and select `End existing session`. + +

+ ThinLinc Login Popup +

+ + Select "End existing session" and try "Connect" again. + +### ThinLinc session unreachable + +#### Problem + +When trying to login to ThinLinc and re-connect to your existing session, you receive an error *"Your ThinLinc session is currently unreachable"*. + +#### Solution + +This can happen if the specific login node your existing remote desktop session was residing on is currently offline or down, so ThinLinc can not reconnect to your existing session. Most often the session is non-recoverable at this point, so the solution is to terminate your existing ThinLinc desktop session and start a new one. + +* **If you are using a web-version ThinLinc remote desktop (inside the browser):** + + The web version does not have the capability to kill the existing session, only the standalone client does. Please install the standalone client and follow the steps below: + + [ThinLinc](accounts.md#thinlinc) + +* **If you are using a ThinLinc client:** + + Close the ThinLinc client, reopen the client login popup, and select `End existing session`. + +

+ ThinLinc Login Popup +

+ + Select "End existing session" and try "Connect" again. + +### How to disable ThinLinc screensaver + +#### Problem + +Your ThinLinc desktop is locked after being idle for a while, and it asks for a password to refresh it. It means the "screensaver" and "lock screen" functions are turned on, but you want to disable these functions. + +#### Solution + +If your screen is locked, close the ThinLinc client, reopen the client login popup, and select `End existing session`. + +

+ ThinLinc Login Popup +

+ +Select "End existing session" and try "Connect" again. + +To permanently avoid screen lock issue, right click desktop and select `Applications`, then `settings`, and select `Screensaver`. + +

+ ThinLinc Screensaver +

+ +Select "Applications", then "settings", and select "Screensaver". + +Under **Screensaver**, turn off the `Enable Screensaver`, then under **Lock Screen**, turn off the `Enable Lock Screen`, and close the window. + +

+ ThinLinc Disable Screensaver +

+ +Under "Screensaver" tab, turn off the "Enable Screensaver" option. + +

+ ThinLinc Disable Lock Screen +

+ +Under "Lock Screen" tab, turn off the "Enable Lock Screen" option. + +### I worked on Negishi after I graduated/left Purdue, but can not access it anymore + +#### Problem + +You have graduated or left Purdue but continue collaboration with your Purdue colleagues. You find that your access to Purdue resources has suddenly stopped and your password is no longer accepted. + +#### Solution + +Access to all resources depends on having a valid Purdue Career Account. Expired Career Accounts are removed twice a year, during Spring and October breaks (more details at the [official page](https://www.purdue.edu/apps/account/IAMO/Purdue_CareerAccount_Expiration.jsp)). If your Career Account was purged due to expiration, you will not be be able to access the resources. + +To provide remote collaborators with valid Purdue credentials, the University provides a special procedure called [Request for Privileges (R4P)](https://www.purdue.edu/apps/account/r4p). If you need to continue your collaboration with your Purdue PI, the PI will have to submit or renew an R4P request on your behalf. + +After your R4P is completed and Career Account is restored, please note two additional necessary steps: + +* **Access:** Restored Career Accounts by default do **not** have any RCAC resources enabled for them. **Your PI will have to login to the [Manage Users](https://www.rcac.purdue.edu/account/groups) tool and explicitly re-enable your access by un-checking and then ticking back checkboxes for desired queues/Unix groups resources.** +* **Email:** Restored Career Accounts by default do **not** have their *@purdue.edu* email service enabled. While this does not preclude you from using RCAC resources, any email messages (be that generated on the clusters, or any service announcements) would not be delivered - which may cause inconvenience or loss of compute jobs. To avoid this, we recommend setting your restored *@purdue.edu* email service to "Forward" (to an actual address you read). The easiest way to ensure it is to go through the [Account Setup process](https://www.purdue.edu/apps/account/AccountSetup). + +## Jobs + +### cannot connect to X server / cannot open display + +#### Problem + +You receive the following message after entering a command to bring up a graphical window + +`cannot connect to X server` `cannot open display` + +#### Solution + +This can happen due to multiple reasons: + +1. Reason: Your SSH client software does not support graphical display by itself (e.g. SecureCRT or PuTTY). + * Solution: Try using a client software like ThinLinc or MobaXterm as described in the [SSH X11 Forwarding guide](accounts.md#ssh-x11-forwarding). +2. Reason: You did not enable X11 forwarding in your SSH connection. + + * Solution: If you are in a Windows environment, make sure that X11 forwarding is enabled in your connection settings (e.g. in MobaXterm or PuTTY). If you are in a Linux environment, try + + `ssh -Y -l username hostname` + +3. Reason: If you are trying to open a graphical window within an interactive PBS job, make sure you are using the `-X` option with `qsub` after following the previous step(s) for connecting to the front-end. Please see the example in the [Interactive Jobs guide](../run_jobs/interactive_jobs). +4. Reason: If none of the above apply, make sure that you are [within quota of your home directory](#usrbinxauth-error-in-locking-authority-file). + +### bash: command not found + +#### Problem + +You receive the following message after typing a command + +`bash: command not found` + +#### Solution + +This means the system doesn't know how to find your command. Typically, you need to load a module to do it. + +### bash: module command not found + +#### Problem + +You receive the following message after typing a command, e.g. module load intel + +`bash: module command not found` + +#### Solution + +The system cannot find the module command. You need to source the modules.sh file as below + +`source /etc/profile.d/modules.sh` + +or + +`#!/bin/bash -i` + +### Close Firefox / Firefox is already running but not responding + +--8<-- "docs/snippets/firefox_lock.md" + +### Jupyter: database is locked / can not load notebook format + +--8<-- "docs/snippets/jupyter_lock.md" + +### How do I know Non-uniform Memory Access (NUMA) layout on Negishi? + +* You can learn about processor layout on Negishi nodes using the following command: + + ``` + a003.negishi:~$ lstopo-no-graphics + ``` + +* For detailed IO connectivity: + + ``` + a003.negishi:~$ lstopo-no-graphics --physical --whole-io + ``` + +* Please note that NUMA information is useful for advanced MPI/OpenMP/GPU optimizations. For most users, using default NUMA settings in MPI or OpenMP would give you the best performance. + +### Why cannot I use --mem=0 when submitting jobs? + +#### Question + +Why can't I specify `--mem=0` for my job? + +#### Answer + +We no longer support requesting unlimited memory (`--mem=0`) as it has an adverse effect on the way scheduler allocates job, and could lead to large amount of nodes being blocked from usage. + +!!! note + Most often we suggest relying on default memory allocation (cluster-specific). But if you have to request custom amounts of memory, you can do it explicitly. For example `--mem=20G`. + +If you want to use the entire node's memory, you can submit the job with the `--exclusive` option. + +### Can I extend the walltime on a job? + +In some circumstances, yes. Walltime extensions must be requested of and completed by staff. Walltime extension requests will be considered on named (your advisor or research lab) queues. **Standby or debug queue jobs cannot be extended**. + +Extension requests are at the discretion of staff based on factors such as any upcoming maintenance or resource availability. Extensions can be made past the normal maximum walltime on named queues but these jobs are subject to early termination should a conflicting maintenance downtime be scheduled. + +Please be mindful of time remaining on your job when making requests and make requests at least 24 hours before the end of your job AND during business hours. We cannot guarantee jobs will be extended in time with less than 24 hours notice, after-hours, during weekends, or on a holiday. + +We ask that you make accurate walltime requests during job submissions. Accurate walltimes will allow the job scheduler to efficiently and quickly schedule jobs on the cluster. Please consider that extensions can impact scheduling efficiency for all users of the cluster. + +Requests can be made by [contacting support](https://www.rcac.purdue.edu/help). We ask that you: + +* Provide numerical job IDs, cluster name, and your desired extension amount. +* Provide at least 24 hours notice before job will end (more if request is made on a weekend or holiday). +* Consider making requests during business hours. We may not be able to respond in time to requests made after-hours, on a weekend, or on a holiday. + +## Data + +### How is my Data Secured on Negishi? + +Negishi is operated in line with policies, standards, and best practices as described within [Secure Purdue](https://www.purdue.edu/securepurdue), and specific to [RCAC Resources](https://www.rcac.purdue.edu/policies). + +Security controls for Negishi are based on ones defined in NIST cybersecurity standards. + +Negishi supports research at the L1 fundamental and L2 sensitive levels. +Negishi is not approved for storing data at the L3 restricted (covered by HIPAA) or L4 Export Controlled (ITAR), or any Controlled Unclassified Information (CUI). + +For resources designed to support research with heightened security requirements, please look for resources within the [REED+ Ecosystem](https://www.rcac.purdue.edu/services/reedplus). + +### Can I share data with outside collaborators? + +Yes! Globus allows convenient sharing of data with outside collaborators. Data can be shared with collaborators' personal computers or directly with many other computing resources at other institutions. See the Globus documentation on how to share data: + +* + +### HSI/HTAR: Unable to authenticate user with remote gateway (error 2 or 9) + +There could be a variety of such errors, with wordings along the lines of + +``` +Could not initialize keytab on remote server. +result = -2, errno = 2rver connection +*** hpssex_OpenConnection: Unable to authenticate user with remote gateway at 128.211.138.40.1217result = -2, errno = 9 +Unable to setup communication to HPSS... +ERROR (main) unable to open remote gateway server connection +HTAR: HTAR FAILED +``` + +and + +``` +*** hpssex_OpenConnection: Unable to authenticate user with remote gateway at 128.211.138.40.1217result = -11000, errno = 9 +Unable to setup communication to HPSS... +*** HSI: error opening logging +Error - authentication/initialization failed +``` + +The root cause for these errors is an expired or non-existent keytab file (a special authentication token stored in your home directory). These keytabs are valid for 90 days and on most RCAC resources they are usually automatically checked and regenerated when you execute `hsi` or `htar` commands. However, if the keytab is invalid, or fails to generate, Fortress may be unable to authenticate you and you would see the above errors. This is especially common on those RCAC clusters that have their own dedicated home directories (such as Negishi), or on standalone installations (such as if you downloaded and installed HSI and HTAR on your non-RCAC computer). + +*This is a temporary problem and a permanent system-wide solution is being developed.* In the interim, the recommended workaround is to generate a new valid keytab file in your main research computing home directory, and then copy it to your home directory on Negishi. The `fortresskey` command is used to generate the keytab and can be executed on another cluster or a dedicated data management host `data.rcac.purdue.edu`: + +``` +$ ssh myusername@data.rcac.purdue.edu fortresskey +$ scp -pr myusername@data.rcac.purdue.edu:~/.private $HOME +``` + +With a valid keytab in place, you should then be able to use `hsi` and `htar` commands to access Fortress from Negishi. Note that only one keytab can be valid at any given time (i.e. if you regenerated it, you may have to copy the new keytab to all systems that you intend to use `hsi` or `htar` from if they do not share the main research computing home directory). + +### HSI/HTAR: put: Error -5 on transfer + +First, check your firewall settings, and ensure that there are no firewall rules interfering with connecting to Fortress. For firewall configuration, please see "[Do I need to do anything to my firewall to access Negishi?](#do-i-need-to-do-anything-to-my-firewall-to-access-negishi)" **If firewalls are not responsible:** + +Open the file named `/etc/hosts` on your workstation, especially if you run a Debian or Ubuntu Linux distribution. Look for a line like: + +``` +127.0.1.1 hostname.dept.purdue.edu hostname +``` + +Replace the IP address 127.0.1.1 with the real IP address for your system. If you don't know your IP address, you can find it with the command: + +``` +host `hostname --fqdn` +``` + +### Can I access Fortress from Negishi? + +Yes. While Fortress directories are not directly mounted on Negishi for performance and archival protection reasons, they can be accessed from Negishi front-ends and nodes using any of the recommended methods of [HSI, HTAR or Globus](https://www.rcac.purdue.edu/knowledge/fortress/storage/transfer). + +## Software + +### Cannot use pip after loading ml-toolkit modules + +#### Question + +Pip throws an error after loading the machine learning modules. How can I fix it? + +#### Answer + +Machine learning modules (tensorflow, pytorch, opencv etc.) include a version of `pip` that is newer than the one installed with Anaconda. As a result it will throw an error when you try to use it. + +``` +$ pip --version +Traceback (most recent call last): + File "/apps/cent7/anaconda/5.1.0-py36/bin/pip", line 7, in + from pip import main +ImportError: cannot import name 'main' +``` + +The preferred way to use `pip` with the machine learning modules is to invoke it via Python as shown below. + +``` +$ python -m pip --version +``` + +### How can I get access to Sentaurus software? + +#### Question + +How can I get access to Sentaurus tools for micro- and nano-electronics design? + +#### Answer + +Sentaurus software license requires a signed NDA. Please contact [Dr. Mark Johnson, Director of ECE Instructional Laboratories](https://engineering.purdue.edu/Mark-Johnson) to complete the process. + +Once the licensing process is complete and you have been added into a `cae2` Unix group, you could use Sentaurus on RCAC community clusters by loading the corresponding environment module: + +``` +module load sentaurus +``` + +### Julia package installation + +Users do not have write permission to the default julia package installation destination. However, users can install packages into home directory under `~/.julia`. + +Users can side step this by explicitly defining where to put julia packages: + +``` +$ export JULIA_DEPOT_PATH=$HOME/.julia +$ julia -e 'using Pkg; Pkg.add("PackageName")' +``` + +## About Research Computing + +### Can I get a private server from RCAC? + +#### Question + +Can I get a private (virtual or physical) server from RCAC? + +#### Answer + +Often, researchers may want a private server to run databases, web servers, or other software. RCAC currently has [Geddes](https://www.rcac.purdue.edu/compute/geddes), a Community Composable Platform optimized for composable, cloud-like workflows that are complementary to the batch applications run on Community Clusters. Funded by the National Science Foundation under grant OAC-2018926, Geddes consists of Dell Compute nodes with two 64-core AMD Epyc 'Rome' processors (128 cores per node). + +To purchase access to Geddes today, go to the [Cluster Access Purchase](https://www.rcac.purdue.edu/purchase) page. Please subscribe to our Community Cluster Program Mailing List to stay informed on the latest purchasing developments or contact us (rcac-cluster-purchase@lists.purdue.edu) if you have any questions. + +[**Back to Negishi User Guide**](index.md) diff --git a/docs/userguides/negishi/gateway.md b/docs/userguides/negishi/gateway.md new file mode 100644 index 00000000..06f53c97 --- /dev/null +++ b/docs/userguides/negishi/gateway.md @@ -0,0 +1,33 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Gateway (Open OnDemand) + +Negishi's Gateway is an [open-source HPC portal](http://openondemand.org/) developed by the [Ohio Supercomputing Center](https://www.osc.edu/). Open OnDemand allows one to interact with HPC resources through a web browser and easily manage files, submit jobs, and interact with graphical applications directly in a browser, all with no software to install. Negishi has an instance of OnDemand available that can be accessed via [gateway.negishi.rcac.purdue.edu](https://gateway.negishi.rcac.purdue.edu). + +## Logging In + +To log into Gateway: + +* Navigate to [gateway.negishi.rcac.purdue.edu](https://gateway.negishi.rcac.purdue.edu) +* Log in using your Career account username and Purdue Login Duo client. + +On the splash page you will see a quota usage report. If you are over 90% on any of your quotas a warning will be displayed. This information will update every 10-15 minutes while you are active on Gateway. + +## Apps + +There are a number of built-in apps in Gateway that can be accessed from the top menu bar. Below are links to documentation on each app. + +- [Interactive Apps](gateway/interactive.md) +- [Files](gateway/files.md) +- [Jobs](gateway/jobs.md) +- [Cluster Tools](gateway/cluster.md) + +[**Back to Negishi User Guide**](index.md) diff --git a/docs/userguides/negishi/gateway/cluster.md b/docs/userguides/negishi/gateway/cluster.md new file mode 100644 index 00000000..8e58ae4e --- /dev/null +++ b/docs/userguides/negishi/gateway/cluster.md @@ -0,0 +1,19 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Cluster Tools + +The Cluster Tools menu contains cluster utilities. At the moment, only a terminal app is provided. Additional apps may be developed and provided in the future. + +## Shell Access + +Launching the shell app will provide you with a web-based terminal session on the cluster front-end. This is equivalent to using a standalone SSH client to connect to `negishi.rcac.purdue.edu` where you are connected to one several front-ends. The normal acceptable [front-end use policy](https://www.rcac.purdue.edu/policies/frontenduse) applies to access through the web-app. X11 Forwarding is not supported. Use of one of the [interactive apps](interactive.md) is recommended for graphical applications. + +[**Back to the Gateway (Open OnDemand) section**](../gateway.md) diff --git a/docs/userguides/negishi/gateway/files.md b/docs/userguides/negishi/gateway/files.md new file mode 100644 index 00000000..c57f47d9 --- /dev/null +++ b/docs/userguides/negishi/gateway/files.md @@ -0,0 +1,41 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Files + +The Files app will let you access your files in your [Home Directory](../storage/home_directory.md), [Scratch](../storage/scratch_space.md), and [Data Depot](https://www.rcac.purdue.edu/storage/depot) spaces. The app lets you manage create, manage, and delete files and directories from your web browser. Navigate by double clicking on folders in the file explorer or by using the file tree on the left. + +

+ Open OnDemand file browser +

+ + +The browser-based file explorer. Navigate by double clicking on folders in the file explorer or by using the file tree on the left. + +On the top row, there are buttons to: + +* Go To: directly input a directory to navigate to +* Open in Terminal: launches the Shell app and navigates you to the current directory in the terminal +* New File: creates a new, empty file +* New Dir: creates a new, empty directory +* Upload: upload a file from your computer + +**Note:** File uploads from your browser are limited to 100 GB per file. Be mindful that uploads over a few gigabytes may be unreliable through your browser, especially from off-campus connections. For very large files or off-campus transfers alternative methods such as [Globus](../storage/globus.md) are highly recommended. + +The second row of buttons lets you perform typical file management operations. The Edit button will open files in a fully fledged browser based text editor - it features syntax highlighting and vim and Emacs key bindings. + +

+ Open OnDemand file editor +

+ + +The browser-based text editor interface, shown here editing a Bash script, includes syntax highlighting, font-size adjustments, and various key bindings. + +[**Back to the Gateway (Open OnDemand) section**](../gateway.md) diff --git a/docs/userguides/negishi/gateway/interactive.md b/docs/userguides/negishi/gateway/interactive.md new file mode 100644 index 00000000..aab329f0 --- /dev/null +++ b/docs/userguides/negishi/gateway/interactive.md @@ -0,0 +1,24 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Interactive Apps + +There are several interactive apps available through Gateway that can be accessed through the Interactive Apps dropdown menu. These apps are provided with a basic node and software configuration as a 'quick-launch' option to get your work up and running quickly. For simplicity, minimal options are provided - these apps are not intended for complex configuration/customization scenarios. + +After you a submit an interactive app to the queue, Gateway will track and manage the session. Once it starts, you may connect and disconnect from the session in your browser, leaving the job running while you log out of your browser. + +Each of the available apps are documented through the following links. + +- [Compute Node Desktop](interactive/desktop.md) +- [Jupyter Notebook](interactive/notebook.md) +- [MATLAB](interactive/matlab.md) +- [RStudio Server](interactive/rstudio.md) + +[**Back to the Gateway (Open OnDemand) section**](../gateway.md) diff --git a/docs/userguides/negishi/gateway/interactive/desktop.md b/docs/userguides/negishi/gateway/interactive/desktop.md new file mode 100644 index 00000000..076bb02b --- /dev/null +++ b/docs/userguides/negishi/gateway/interactive/desktop.md @@ -0,0 +1,21 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Compute Node Desktop + +The Compute Node Desktop app will launch a graphical desktop session on a compute node. This is similar to using [ThinLinc](/userguides/negishi/accounts/#thinlinc), however, this gives you a desktop directly on a compute node instead on a front-end. This app is useful if you have a custom application or application not directly available as an interactive app you would like to run inside Gateway. + +To launch a desktop session on a compute node, select the Negishi Compute Desktop app. From the submit form, select from the available options - the queue to which you wish to submit and the number of wallclock hours you wish to have job running. There is also a checkbox that enable a notification to your email when the job starts. + +After the interactive job is submitted you will be taken to your list of active interactive app sessions. You can monitor the status of the job from here until it starts, or if you enabled the email notification, watch your Purdue email for the notification the job has started. + +Once it is indicated the job has started you can connect to the desktop with the "Launch noVNC in New Tab" button. The session will be terminated after the wallclock hours you specified have elapsed or you terminate the session early with the "Delete" button from the list of sessions. Deleting the session when you are finished will free up queue resources for your lab mates and other users on the system. + +[**Back to the Interactive Apps section**](../interactive.md) diff --git a/docs/userguides/negishi/gateway/interactive/matlab.md b/docs/userguides/negishi/gateway/interactive/matlab.md new file mode 100644 index 00000000..6041469e --- /dev/null +++ b/docs/userguides/negishi/gateway/interactive/matlab.md @@ -0,0 +1,24 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# MATLAB + +The MATLAB app will launch a MATLAB session on a compute node and allow you to connect directly to it in a web browser. + +To launch a MATLAB session on a compute node, select the MATLAB app. From the submit form, select from the available options - the version of MATLAB you are interested in running, the queue to which you wish to submit, and the number of wallclock hours you wish to have job running. There is also a checkbox that enable a notification to your email when the job starts. + +After the interactive job is submitted you will be taken to your list of active interactive app sessions. You can monitor the status of the job from here until it starts, or if you enabled the email notification, watch your Purdue email for the notification the job has started. + +Once it is indicated the job has started you can connect to the desktop with the "Launch noVNC in New Tab" button. The session will be terminated after the wallclock hours you specified have elapsed or you terminate the session early with the "Delete" button from the list of sessions. Deleting the session when you are finished will free up queue resources for your lab mates and other users on the system. + +!!! Warning + There are known issues with running Matlab in this way and resizing your web browser. Graphical corruption may occur if you resize the browser. Fixes for this are being investigated. + +[**Back to the Interactive Apps section**](../interactive.md) diff --git a/docs/userguides/negishi/gateway/interactive/notebook.md b/docs/userguides/negishi/gateway/interactive/notebook.md new file mode 100644 index 00000000..d23ce9b8 --- /dev/null +++ b/docs/userguides/negishi/gateway/interactive/notebook.md @@ -0,0 +1,31 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Jupyter Notebook + +The Notebook app will launch a Notebook session on a compute node and allow you to connect directly to it in a web browser. + +To launch a Notebook session on a compute node, select the Notebook app. From the submit form, select from the available options: + +1. **Queue:**This is a dropdown menu from which you can select a queue from all of the queues to which you have permission to submit. +2. **Walltime:**This is a field which expects a number and represents how many hours you want to keep the session running. Note that this value should not exceed the maximum value given next to the selected queue name from the queue dropdown menu. +3. **Number of Cores/GPUs:** This is a field which expects a number and represents the number of your resources your session is requesting. Note that the amount of memory allocated for your session is proportional to the number of cores or GPUs that you request for your job, so if your session is running out of memory, consider increasing this value. +4. **Use Jupyter Lab:**This is a checkbox which, when checked, will run Jupyter Lab instead of Jupyter Notebook. Both of these applications are interfaces to Jupyter, and you can launch Jupyter notebooks from within Jupyter Lab. Jupyter Notebook is more "barebones" while Jupyter Lab has additional features such as the ability to interact with additional file types. +5. **E-mail Notice:**This is a checkbox which, when checked, will send you an e-mail notification to your Purdue e-mail that your session is ready when the scheduler has found resources to dedicate to your session. + +After the interactive job is submitted you will be taken to your list of active interactive app sessions. You can monitor the status of the job from here until it starts, or if you enabled the email notification, watch your Purdue email for the notification the job has started. + +Once it is indicated the job has started you can connect to the desktop with the "Connect to Jupyter" button. Once connected, you can create new notebooks, selecting the currently available Anaconda versions available as modules, and any personally created Notebook kernels. + +Often times you may want to use one of your existing Anaconda environments within your Jupyter session to use libraries specific to your workflow. In order to do so, you must ensure that the Anaconda environment you want to use contains the Python packages "IPyKernel" and "IPython" which are packages that are required by Jupyter. When you create a Jupyter session, Open OnDemand will check through your existing Anaconda environments and create a Jupyter kernel for any Anaconda environment that contains these two packages, and you will be able to select to use that kernel from within the application. + +The session will be terminated after the number of hours you specified have elapsed or you terminate the session early with the "Delete" button from the list of sessions. Deleting the session when you are finished will free up queue resources for your lab mates and other users on the system. + +[**Back to the Interactive Apps section**](../interactive.md) diff --git a/docs/userguides/negishi/gateway/interactive/rstudio.md b/docs/userguides/negishi/gateway/interactive/rstudio.md new file mode 100644 index 00000000..b8a1f90a --- /dev/null +++ b/docs/userguides/negishi/gateway/interactive/rstudio.md @@ -0,0 +1,21 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# RStudio Server + +The RStudio app will launch a RStudio session on a compute node and allow you to connect directly to it in a web browser. + +To launch a RStudio session on a compute node, select the RStudio app. From the submit form, select from the available options - the queue to which you wish to submit, and the number of wallclock hours you wish to have job running. There is also a checkbox that enable a notification to your email when the job starts. + +After the interactive job is submitted you will be taken to your list of active interactive app sessions. You can monitor the status of the job from here until it starts, or if you enabled the email notification, watch your Purdue email for the notification the job has started. + +Once it is indicated the job has started you can connect to the desktop with the "Connect to RStudio Server" button. The session will be terminated after the wallclock hours you specified have elapsed or you terminate the session early with the "Delete" button from the list of sessions. Deleting the session when you are finished will free up queue resources for your lab mates and other users on the system. + +[**Back to the Interactive Apps section**](../interactive.md) diff --git a/docs/userguides/negishi/gateway/jobs.md b/docs/userguides/negishi/gateway/jobs.md new file mode 100644 index 00000000..2a80a95c --- /dev/null +++ b/docs/userguides/negishi/gateway/jobs.md @@ -0,0 +1,102 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Jobs + +There are two apps under the Jobs apps: Active Jobs and Job Composer. These are detailed below. + +## Active Jobs + +This shows you active SLURM jobs currently on the cluster. The default view will show you your current jobs, similar to `squeue -u rices`. Using the button labeled "Your Jobs" in the upper right allows you to select different filters by queue (account). All accounts output by `slist` will appear for you here. Using the arrow on the left hand side will expand the full job details. + + +

+ A table of active jobs +

+ + +The table of active jobs shows useful information such as queue, status, cluster, and ID. It can be sorted by clicking the headers of each column or searched with the "Filter" box above it. + + +### Job Composer + +The Job Composer app allows you to create and submit jobs to the cluster. You can select from pre-defined templates (most of these are taken from the User Guide examples) or you can create your own templates for frequently used workflows. + +### Creating Job from Existing Template + +Click "New Job" menu, then select "From Template": + +

+ The job composer interface +

+ + +When clicking the 'New Job' button a drop-down will show a few options. "From Template" is usually the second item in the list. + +Then select from one of the available templates. + + +

+ A sortable data table containing a list of all the available templates. +

+ + +Select one of the templates by clicking its row in the table of available templates. + + +Click 'Create New Job' in second pane. + + +

+ The 'Create New Job' pane +

+ + +The "Create New Job" pane will show form options for "Job Name", "Cluster", and "Script Name" with the "Create New Job" button below. + + +Your new job should be selected in your list of jobs. In the 'Submit Script' pane you can see the job script that was generated with an 'Open Editor' link to open the script in the built-in editor. Open the file in the editor and edit the script as necessary. By default the job will specify standby queue - this should be changed as appropriate, along with the node and walltime requests. + + +

+ The 'Submit Script' pane +

+ + +The "Submit Script" pane will show a preview of the contents of the script file and action buttons below. + + +When you are finished with editing the job and are ready to submit, click the green 'Submit' button at the top of the job list. You can monitor progress from here or from the Active Jobs app. Once completed, you should see the output files appear: + +

+ A list of files found in the output folder +

+ + +The folder contents will be listed, showing the resulting output files from running the submitted script. + +Clicking on one of the output files will open it in the file editor for your viewing. + +## Creating New Template + +First, prepare a template directory containing a template submission script along with any input files. Then, to import the job into the Job Composer app, click the 'Create New Template' button. Fill in the directory containing your template job script and files in the first box. Give it an appropriate name and notes. + + +

+ The 'Create New Template' form +

+ + +The "Create New Template" form has inputs for "Path", "Name", "Cluster", and "Notes". If "Path" is left blank, a default job script will be added to the new template. + + +This template will now appear in your list of templates to choose from when composing jobs. You can now go create and submit a job from this new template. + +[**Back to the Gateway (Open OnDemand) section**](../gateway.md) diff --git a/docs/userguides/negishi/index.md b/docs/userguides/negishi/index.md new file mode 100644 index 00000000..a5779cfe --- /dev/null +++ b/docs/userguides/negishi/index.md @@ -0,0 +1,23 @@ +--- +#tags: +# - Negishi +authors: + - hkashgar +search: + boost: 2 +--- + +# Negishi User Guide + +Negishi is a Community Cluster optimized for communities running traditional, tightly-coupled science and engineering applications. + +- [**Negishi Overview**](overview.md) +- [**Biography of Negishi**](biography.md) +- [**Accounts**](accounts.md) +- [**Software**](software.md) +- [**Running Jobs**](run_jobs/index.md) +- [**File Storage and Transfer**](storage.md) +- [**Gateway (Open OnDemand)**](gateway.md) +- [**Compiling Source Code**](compile.md) +- [**Frequently Asked Questions**](faqs.md) + diff --git a/docs/userguides/negishi/overview.md b/docs/userguides/negishi/overview.md new file mode 100644 index 00000000..4366f8ec --- /dev/null +++ b/docs/userguides/negishi/overview.md @@ -0,0 +1,51 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Negishi Overview + +Negishi is a Community Cluster optimized for communities running traditional, tightly-coupled science and engineering applications. Negishi is being built through a partnership with Dell and AMD over the summer of 2022. Negishi consists of Dell compute nodes with two 64-core AMD Epyc "Milan" processors (128 cores per node) and 256 GB of memory. All nodes have 100 Gbps HDR Infiniband interconnect and a 6-year warranty. + +New with Negishi is that access is being offered on the basis of each 64-core Rome processor, or a half-node share. To purchase access to Negishi today, go to the [Cluster Access Purchase](https://www.rcac.purdue.edu/purchase) page. Please subscribe to our Community Cluster Program Mailing List to stay informed on the latest purchasing developments or contact us via email at [rcac-cluster-purchase@lists.purdue.edu](mailto:rcac-cluster-purchase@lists.purdue.edu) if you have any questions. + +## Negishi Interactive + +The interactive tier on our Negishi cluster provides entry-level access to high performance computing. This includes login to the system, data storage on our high-performance *scratch* filesystem, and a small allocation that allows jobs submitted to an "interactive" account limited to a few cores. This subscription is useful for getting workloads off your personal machine, integrated with more robust research computing and data systems, and a platform for smaller workloads. Transitioning to a larger allocation with priority scheduling is easy and simple. + +## Negishi Namesake + +Negishi is named in honor of Dr. Ei-ichi Negishi, the Herbert C. Brown Distinguished Professor in the Department of Chemistry at Purdue. More information about his life and impact on Purdue is available in a [Biography of Negishi](./biography.md). + +## Negishi Specifications + +All Negishi compute nodes have 128 processor cores, 256 GB memory and 100 Gbps HDR100 Infiniband interconnects. + +### Negishi Front-Ends + +| Front-Ends | Number of Nodes | Processors per Node | Cores per Node | Memory per Node | Retires in | +| --- | --- | --- | --- | --- | --- | +| | 8 | Two AMD EPYC 7763 64-Core Processors @ 2.2GHz | 128 | 512 GB | 2028 | + +### Negishi Sub-Clusters + +| Sub-Cluster | Number of Nodes | Processors per Node | Cores per Node | Memory per Node | Retires in | +| --- | --- | --- | --- | --- | --- | +| A | 450 | Two AMD Epyc 7763 “Milan” CPUs @ 2.2GHz | 128 | 256 GB | 2028 | +| B | 6 | Two AMD Epyc 7763 “Milan” CPUs @ 2.2GHz | 128 | 1 TB | 2028 | +| C | 16 | Two AMD Epyc 7763 “Milan” CPUs @ 2.2GHz | 128 | 512 GB | 2028 | +| G | 5 | Two AMD Epyc 7313 “Milan” CPUs @ 3.0GHz, Three AMD MI210 GPUs (64GB) | 32 | 512 GB | 2028 | + +Negishi nodes run Rocky Linux 8 and use Slurm (Simple Linux Utility for Resource Management) as the batch scheduler for resource and job management. The application of operating system patches occurs as security needs dictate. All nodes allow for unlimited stack usage, as well as unlimited core dump size (though disk space and server quotas may still be a limiting factor). + +On Negishi, the following set of compiler and message-passing libraries for parallel code are recommended: + +* GCC 12.2.0 +* OpenMPI or MVAPICH2 + + diff --git a/docs/userguides/negishi/run_jobs/ansysfluent.md b/docs/userguides/negishi/run_jobs/ansysfluent.md new file mode 100644 index 00000000..3eacb73c --- /dev/null +++ b/docs/userguides/negishi/run_jobs/ansysfluent.md @@ -0,0 +1,57 @@ +--- +tags: + - Negishi +authors: + - jin456 + - remender + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Ansys Fluent + +[Ansys](https://www.ansys.com) is a CAE/multiphysics engineering simulation software that utilizes finite element analysis for numerically solving a wide variety of mechanical problems. The software contains a list of packages and can simulate many structural properties such as strength, toughness, elasticity, thermal expansion, fluid dynamics as well as acoustic and electromagnetic attributes. + +## Ansys Licensing + +The Ansys licensing on our community clusters is maintained by Purdue ECN group. There are two types of licenses: teaching and research. For more information, please refer to [ECN Ansys licensing page](https://engineering.purdue.edu/ECN/Support/KB/Docs/ANSYSFLUENTLicensing). If you are interested in purchasing your own research license, please send email to **software@ecn.purdue.edu**. + +## Ansys Workflow + +Ansys software consists of several sub-packages such as Workbench and Fluent. Most simulations are performed using the Ansys Workbench console, a GUI interface to manage and edit the simulation workflow. It requires X11 forwarding for remote display so a SSH client software with X11 support or a remote desktop portal is required. Please see [Logging In](../accounts.md) section for more details. To ensure preferred performance, [ThinLinc](../accounts.md#thinlinc) remote desktop connection is highly recommended. + +Typically users break down larger structures into small components in geometry with each of them modeled and tested individually. A user may start by defining the dimensions of an object, adding weight, pressure, temperature, and other physical properties. + +Ansys Fluent is a computational fluid dynamics (CFD) simulation software known for its advanced physics modeling capabilities and accuracy. Fluent offers unparalleled analysis capabilities and provides all the tools needed to design and optimize new equipment and to troubleshoot existing installations. + + + + + +## Loading Ansys Module + +Different versions of Ansys are installed on the clusters and can be listed with `module spider` or `module avail` command in the terminal. + +```bash +$ module avail ansys/ +---------------------- Core Applications ----------------------------- + ansys/2019R3 ansys/2020R1 ansys/2021R2 ansys/2022R1 (D) +``` + +Before launching Ansys Workbench, a specific version of Ansys module needs to be loaded. For example, you can `module load ansys/2021R2` to use the latest Ansys 2021R2. If no version is specified, the default module -> (D) (`ansys/2022R1` in this case) will be loaded. You can also check the loaded modules with `module list` command. + +## Launching Ansys Workbench + +Open a terminal on Negishi, enter `rcac-runwb2` to launch Ansys Workbench. + +You can also use `runwb2` to launch Ansys Workbench. The main difference between `runwb2`and `rcac-runwb2` is that the latter sets the project folder to be in your scratch space. Ansys has an known bug that it might crash when the project folder is set to `$HOME` on our systems. + +* [**Preparing Case Files for Fluent**](ansysfluent/preparing_cases.md) +* [**Case Calculating with Fluent**](ansysfluent/calculating.md) +* [**Fluent Text User Interface and Journal File**](ansysfluent/tui_journal.md) +* [**Submitting Fluent jobs to SLURM**](ansysfluent/submit_jobs.md) + + +[**Back to the Running Jobs section**](index.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/ansysfluent/calculating.md b/docs/userguides/negishi/run_jobs/ansysfluent/calculating.md new file mode 100644 index 00000000..89946012 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/ansysfluent/calculating.md @@ -0,0 +1,65 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Case Calculating with Fluent + +### Calculation with Fluent + +Now all the files are ready for the Fluent calculations. Both “Geometry” and “Mesh” cells should have green checks. We can set up the CFD simulation parameters in the Ansys Fluent by double-clicking the “Setup” cell. + +Ansys Fluent Launcher can be started by selecting “editing” on the “Setup” cell with many startup options (e.g. Precision, Parallel, Display). Note that “Dimension” is fixed to “3D” because we are using a 3D model in this project. + +![Ansys Fluent Launcher options](../../../../assets/images/userguides/examples/ansys4.png) + +Ansys Fluent Launcher options. + +After the Fluent is opened, an Ansys Fluent settings file `FFF.set` is written under the folder `$Ansys_PROJECT_FOLDER/elbow_demo_file/dp0/FFF/Fluent/`. + +Then we are going to set up all the necessary parameters for Fluent computation. Here are the key steps for the setup: + +1. Setting up the domain: + * Change the units for length to be consistent with the Mesh; + * Check the mesh statistics and quality; +2. Setting up physics: + * Solver: “Energy”, “Viscous Model”, “Near-Wall Treatment”; + * Materials; + * Zones; + * Boundaries: Inlet, Outlet, Internal, Symmetry, Wall; +3. Solving: + * Solution Methods; + * Reports; + * Initialization; + * Iterations and output frequency. + +Then the calculation will be carried out and the results will be written out into `FFF-1.cas.gz` under folder `$Ansys_PROJECT_FOLDER/elbow_demo_file/dp0/FFF/Fluent/`. + +This file contains all the settings and simulation results which can be loaded for post analysis and re-computation (more details will be introduced in the following sections). If only configurations and settings within the Fluent are needed, we can open independent Fluent or submit Fluent jobs with bash commands by loading the existing case in order to facilitate the computation process. + +Parameters used in demo case (use default if not assigned): + +1. Domain Setup: Length Units=”mm”; +2. Solver: Energy=”on”; Viscous Model=”k-epsilon”; Near-Wall Treatment=”Enhanced Wall Treatment”; +3. Materials: water (Density=1000[kg/m^3]; Specific Heat=4216[J/kg-k]; Thermal Conductivity=0.677[w/m-k]; Viscosity=8e-4[kg/m-s]); +4. Zones=”fluid (water)”; +5. Inlet=”velocity-inlet-large” (Velocity Magnitude=0.4m/s, Specification Method=”Intensity and Hydraulic Diameter”, Turbulent Intensity=5%; Hydraulic Diameter=100mm; Thermal Temperature=293.15k) &”velocity-inlet-small” (Velocity Magnitude=1.2m/s, Specification Method=”Intensity and Hydraulic Diameter”, Turbulent Intensity=5%; Hydraulic Diameter=25mm; Thermal Temperature=313.15k); Internal=”interior-fluid”; Symmetry=”symmetry”; Wall=”wall-fluid”; +6. Solution Methods: Gradient=”Green-Gauss Node Based”; +7. Report: plot residual and “Facet Maximum” for “pressure-outlet” +8. Hybrid Initialization; +9. 300 iterations. + +### Results analysis + +The best methods to view and analyze the simulation should be the Ansys Fluent (directly after computation) or the Ansys CFD-Post (entering “Results” in Ansys Workbench). Both methods are straightforward so we will not cover this part in this tutorial. Here is a final simulation result showing the temperature of the symmetry after 300 iterations for reference: + +![Simulated temperature](../../../../assets/images/userguides/examples/ansys5.png) + +Simulated temperature profile of the symmetry. + +[**Back to Ansys Fluent**](../ansysfluent.md) diff --git a/docs/userguides/negishi/run_jobs/ansysfluent/preparing_cases.md b/docs/userguides/negishi/run_jobs/ansysfluent/preparing_cases.md new file mode 100644 index 00000000..d9a2fbe0 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/ansysfluent/preparing_cases.md @@ -0,0 +1,122 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Preparing Case Files for Fluent + +### Creating a Fluent fluid analysis system + +In the Ansys Workbench, create a new fluid flow analysis by double-clicking the Fluid Flow (Fluent) option under the Analysis Systems in the Toolbox on the left panel. You can also drag-and-drop the analysis system into the Project Schematic. A green dotted outline indicating a potential location for the new system initially appears in the Project Schematic. When you drag the system to one of the outlines, it turns into a red box to indicate the chosen location of the new system. + +![Ansys Workbench GUI](../../../../assets/images/userguides/examples/ansys1.png) + +Ansys Workbench GUI and the Fluid Flow system for Fluent. + +The red rectangle indicates the Fluid Flow system for Fluent, which includes all the essential workflows from “2 Geometry” to “6 Results”. You can rename it and carry out the necessary step-by-step procedures by double-clicking the corresponding cells. + +It is important to save the project. Ansys Workbench saves the project with a `.wbpj` extension and also all the supporting files into a folder with the same name. In this case, a file named `elbow_demo.wbpj` and a folder `$Ansys_PROJECT_FOLDER/elbow_demo_files/` are created in the Ansys project folder: + +```bash +$ ll +total 33 +drwxr-xr-x 7 username itap 9 Mar 3 17:47 elbow_demo_files +-rw-r--r-- 1 username itap 42597 Mar 3 17:47 elbow_demo.wbpj +``` + +You should always “Update Project” and save it after finishing a procedure. + +### Creating Geometry in the Ansys DesignModeler + +Create a geometry in the Ansys DesignModeler (by double-clicking “Geometry” cell in workflow), or import the appropriate geometry file (by right-clicking the Geometry cell and selecting “Import Geometry” option from the context menu). + +You can use Ansys DesignModeler to create 2D/3D geometries or even draw the objects yourself. In our example, we created only half of the elbow pipe because the symmetry of the structure is taken into account to reduce the computation intensity. + +![DesignModeler](../../../../assets/images/userguides/examples/ansys2.png) + +Elbow pipe created in Ansys DesignModeler. + +After saving the geometry, a geometry file `FFF.agdb` will be created in the folder: `$Ansys_PROJECT_FOLDER/elbow_demo_file/dp0/FFF/DM/`. The project in Workbench will be updated automatically. + +If you import a pre-existing geometry into Ansys DesignModeler, it will also generate this file with the same filename at this location. + +### Creating mesh in the Ansys Meshing + +Now that we have created the elbow pipe geometry, a computational mesh can be generated by the Meshing application throughout the flow volume. + +With the successful creation of the geometry, there should be a green check showing the completion of “Geometry” in the Ansys Workbench. A Refresh Required icon within the “Mesh” cell indicates the mesh needs to be updated and refreshed for the system. + +![AnsysWorkbenchCells](../../../../assets/images/userguides/examples/ansys3.png) + +Status for different cells shown in Ansys Workbench. + +Then it’s time to open the Ansys Meshing application by double-clicking the “Mesh” cell and editing the mesh for the project. Generally, there are several steps we need to take to define the mesh: + +1. Create names for all geometry boundaries such as the inlets, outlets and fluid body. Note: You can use the strings “velocity inlet” and “pressure outlet” in the named selections (with or without hyphens or underscore characters) to allow Ansys Fluent to automatically detect and assign the corresponding boundary types accordingly. Use “Fluid” for the body to let Ansys Fluent automatically detect that the volume is a fluid zone and treat it accordingly. +2. Set basic meshing parameters for the Ansys Meshing application. Here are several important parameters you may need to assign: Sizing, Quality, Body Sizing Control, Inflation. +3. Select “Generate” to generate the mesh and “Update” to update the mesh into the system. Note: Once the mesh is generated, you can view the mesh statistics by opening the Statistics node in the Details of “Mesh” view. This will display information such as the number of nodes and the number of elements, which gives you a general idea for the future computational resources and time. + +After generation and updating the mesh, a mesh file `FFF.msh` will be generated in folder `$Ansys_PROJECT_FOLDER/elbow_demo_file/dp0/FFF/MECH/` and a mesh database file `FFF.mshdb` will be generated in folder `$Ansys_PROJECT_FOLDER/elbow_demo_file/dp0/global/MECH/`. + +Parameters used in demo case (use default if not assigned): + +1. Length Unit=”mm” +2. Names defined for geometry: + * velocity-inlet-large (large inlet on pipe); + * velocity-inlet-small (small inlet on pipe); + * pressure-outlet (outlet on pipe); + * symmetry (symmetry surface); + * Fluid (body); +3. Mesh: + * Quality: Smoothing=”high”; + * Inflation: Use Automatic Inflation=“Program Controlled”, Inflation Option=”Smooth Transition”; +4. Statistics: + * Nodes=29371; + * Elements=87647. + +### Calculation with Fluent + +Now all the preparations have been ready for the numerical calculation in Ansys Fluent. Both “Geometry” and “Mesh” cells should have green checks on. We can set up the CFD simulation parameters in Ansys Fluent by double-clicking the “Setup” cell. + +When Ansys Fluent is first started or by selecting “editing” on the “Setup” cell, the Fluent Launcher is displayed, enabling you to view and/or set certain Ansys Fluent start-up options (e.g. Precision, Parallel, Display). Note that “Dimension” is fixed to “3D” because we are using a 3D model in this project. + +After the Fluent is opened, an Ansys Fluent settings file `FFF.set` is written under the folder `$Ansys_PROJECT_FOLDER/elbow_demo_file/dp0/FFF/Fluent/`. + +Then we are going to set up all the necessary parameters for Fluent computation. Here are the key steps for the setup: + +1. Setting up the domain: + * Change the units for length to be consistent with the Mesh; + * Check the mesh statistics and quality; +2. Setting up physics: + * Solver: “Energy”, “Viscous Model”, “Near-Wall Treatment”; + * Materials; + * Zones; + * Boundaries: Inlet, Outlet, Internal, Symmetry, Wall; +3. Solving: + * Solution Methods; + * Reports; + * Initialization; + * Iterations and output frequency. + +Then the calculation will be carried out and the results will be written out into `FFF-1.cas.gz` under folder `$Ansys_PROJECT_FOLDER/elbow_demo_file/dp0/FFF/Fluent/`. + +This file contains all the settings and simulation results which can be loaded for post analysis and re-computation (more details will be introduced in the following sections). If only configurations and settings within the Fluent are needed, we can open independent Fluent or submit Fluent jobs with bash commands by loading the existing case in order to facilitate the computation process. + +Parameters used in demo case (use default if not assigned): + +1. Domain Setup: Length Units=”mm”; +2. Solver: Energy=”on”; Viscous Model=”k-epsilon”; Near-Wall Treatment=”Enhanced Wall Treatment”; +3. Materials: water (Density=1000[kg/m^3]; Specific Heat=4216[J/kg-k]; Thermal Conductivity=0.677[w/m-k]; Viscosity=8e-4[kg/m-s]); +4. Zones=”fluid (water)”; +5. Inlet=”velocity-inlet-large” (Velocity Magnitude=0.4m/s, Specification Method=”Intensity and Hydraulic Diameter”, Turbulent Intensity=5%; Hydraulic Diameter=100mm; Thermal Temperature=293.15k) &”velocity-inlet-small” (Velocity Magnitude=1.2m/s, Specification Method=”Intensity and Hydraulic Diameter”, Turbulent Intensity=5%; Hydraulic Diameter=25mm; Thermal Temperature=313.15k); Internal=”interior-fluid”; Symmetry=”symmetry”; Wall=”wall-fluid”; +6. Solution Methods: Gradient=”Green-Gauss Node Based”; +7. Report: plot residual and “Facet Maximum” for “pressure-outlet” +8. Hybrid Initialization; +9. 300 iterations. + +[**Back to Ansys Fluent**](../ansysfluent.md) diff --git a/docs/userguides/negishi/run_jobs/ansysfluent/submit_jobs.md b/docs/userguides/negishi/run_jobs/ansysfluent/submit_jobs.md new file mode 100644 index 00000000..5025ea65 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/ansysfluent/submit_jobs.md @@ -0,0 +1,37 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Submitting Fluent jobs to SLURM + +The Fluent simulations can also run in batch. In this section we provide an example script for submitting Fluent jobs to the SLURM scheduler. Please refer to the [Running Jobs](../../run_jobs/index.md) section of our user guide for detailed tutorials of submitting jobs. + +```bash +#!/bin/bash +# Job script for submitting a FLUENT job on multiple cores on a single node + +# Apply resources via SLURM +#SBATCH --nodes=1 +#SBATCH --ntasks=4 +#SBATCH --time=01:00:00 +#SBATCH --job-name=fluent_test +#SBATCH -o fluent_test_%j.out +#SBATCH -e fluent_test_%j.err + +# Loads Ansys and sets the application up +module purge +module load ansys/2022R1 + +#Initiating Fluent and reading input journal file +fluent 3ddp -t$NTASKS -g -i testJournal.jou +``` + +For more information about submitting Fluent jobs, please refer to [Fluent FAQ]( https://www.cfd-online.com/Wiki/Fluent_FAQ) . + +[**Back to Ansys Fluent**](../ansysfluent.md) diff --git a/docs/userguides/negishi/run_jobs/ansysfluent/tui_journal.md b/docs/userguides/negishi/run_jobs/ansysfluent/tui_journal.md new file mode 100644 index 00000000..c05de7f8 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/ansysfluent/tui_journal.md @@ -0,0 +1,110 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Fluent Text User Interface and Journal File + +### Fluent Text User Interface (TUI) + +If you pay attention to the “Console” window in the Fluent window when setting up and carrying out the calculation, corresponding commands can be found and executed one after another. Almost all the setting processes can be accomplished by the command lines, which is called Fluent Text User Interface (TUI). Here are the main commands in Fluent TUI: + +``` + adjoint/ parallel/ solve/ + define/ plot/ surface/ + display/ preferences/ turbo-workflow/ + exit print-license-usage views/ + file/ report/ + mesh/ server/ +``` + +For example, instead of opening a case by clicking buttons in Ansys Fluent, we can type `/file read-case case_file_name.cas.gz` to open the saved case. + +### Fluent Journal Files + +A Fluent journal file is a series of TUI commands stored in a text file. The file can be written in a text editor or generated by Fluent as a transcript of the commands given to Fluent during your session. + +A journal file generated by Fluent will include any GUI operations (in a TUI form, though). This is quite useful if you have a series of tasks that you need to execute, as it provides a shortcut. To record a journal file, start recording with File -> Write -> Start Journal..., perform whatever tasks you need, and then stop recording with File -> Write -> Stop Journal... + +You can also write your own journal file into a text file. The basic rule for a Fluent journal file is to reproduce the TUI commands that controlled the configuration and calculation of Fluent in their order. You can add a comment in a line starting with a `;` (semicolon). + +Here are some reasons why you should use a Fluent journal file: + +1. Using journal files with bash scripting can allow you to automate your jobs. +2. Using journal files can allow you to parameterize your models easily and automatically. +3. Using a journal file can set parameters you do not have in your case file e.g. autosaving. +4. Using a journal file can allow you to safely save, stop and restart your jobs easily. + +The order of your journal file commands is **highly important**. The correct sequences must be followed and some stages have multiple options e.g. different initialization methods. + +Here is a sample Fluent journal file for the demo case: + +``` + ;testJournal.jou + ;Set the TUI version for Fluent + /file/set-tui-version "22.1" + ;Read the case. The default folder + /file read-case /home/jin456/Fluent_files/tutorial_case1/elbow_files/dp0/FFF/Fluent/FFF-1.cas.gz + ;Initialize the case with Hybrid Initialization + /solve/initialize/hyb-initialization + ;Set Number of Iterations to 1000, Reporting Interval to 10 iterations and Profile Update Interval to 1 iteration + /solve/iterate 1000 10 1 + ;Outputting solver performance data upon completion of the simulation + /parallel timer usage + ;Write out the simulation results. + /file write-case-data /home/jin456/Fluent_files/tutorial_case1/elbow_files/dp0/FFF/Fluent/result.cas.h5 + ;After computation, exit Flent + /exit +``` + +Before running this Fluent journal file, you need to make sure: + +1) the ansys module has been loaded (it’s highly recommended to load the same version of Ansys when you built the case project); + +2) the project case file (`***.cas.gz`) has been created. + +Then we can use Fluent to run this journal file by simply using:`fluent 3ddp -t$NTASKS -g -i testJournal.jou` in the terminal. Here, `3d` indicates this is a 3d model, `dp` indicates double precision, `-t$NTASKS` tells Fluent how many Solver Processes it will take (e.g. `-t4`), `-g` means to run without the GUI or graphics, `-i` testJournal.jou tells Fluent to read the specific journal file. + +Here is a table for the available command line Options for Linux/UNIX and Windows Platforms in Ansys Fluent. + +Options for Fluent TUI + +| Option | Platform | Description | +| --- | --- | --- | +| `-cc` | all | Use the classic color scheme | +| `-ccp x` | Windows only | Use the Microsoft Job Scheduler where x is the head node name. | +| `-cnf=x` | all | Specify the hosts or machine list file | +| `-driver` | all | Sets the graphics driver (available drivers vary by platform - opengl or x11 or null(Linux/UNIX) - opengl or msw or null (Windows)) | +| `-env` | all | Show environment variables | +| `-fgw` | all | Disables the embedded graphics | +| `-g` | all | Run without the GUI or graphics (Linux/UNIX); Run with the GUI minimized (Windows) | +| `-gr` | all | Run without graphics | +| `-gu` | all | Run without the GUI but with graphics (Linux/UNIX); Run with the GUI minimized but with graphics (Windows) | +| `-help` | all | Display command line options | +| `-hidden` | Windows only | Run in batch mode | +| `-host_ip=host:ip` | all | Specify the IP interface to be used by the host process | +| `-i journal` | all | Reads the specified journal file | +| `-lsf` | Linux/UNIX only | Run FLUENT using LSF | +| `-mpi=` | all | Specify MPI implementation | +| `-mpitest` | all | Will launch an MPI program to collect network performance data | +| `-nm` | all | Do not display mesh after reading | +| `-pcheck` | Linux/UNIX only | Checks all nodes | +| `-post` | all | Run the FLUENT post-processing-only executable | +| `-p` | all | Choose the interconnect = default or myr or inf | +| `-r` | all | List all releases installed | +| `-rx` | all | Specify release number | +| `-sge` | Linux/UNIX only | Run FLUENT under Sun Grid Engine | +| `-sge queue` | Linux/UNIX only | Name of the queue for a given computing grid | +| `-sgeckpt ckpt_obj` | Linux/UNIX only | Set checkpointing object to ckpt\_objfor SGE | +| `-sgepe fluent_pe min_n-max_n` | Linux/UNIX only | Set the parallel environment for SGE to fluent\_pe, min\_nand max\_n are number of min and max nodes requested | +| `-tx` | all | Specify the number of processors x | + +For more information for Fluent text user interface and journal files, please refer to [Fluent FAQ]( https://www.cfd-online.com/Wiki/Fluent_FAQ). + + +[**Back to Ansys Fluent**](../ansysfluent.md) diff --git a/docs/userguides/negishi/run_jobs/apptainer.md b/docs/userguides/negishi/run_jobs/apptainer.md new file mode 100644 index 00000000..abc186e5 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/apptainer.md @@ -0,0 +1,123 @@ +--- +tags: + - Negishi + - Apptainer +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Apptainer on Negishi Cluster + +!!! note + Note: Apptainer was formerly known as Singularity and is now a part of the [Linux Foundation](https://apptainer.org/news/community-announcement-20211130). When migrating from Singularity see the [user compatibility documentation](https://apptainer.org/docs/user/main/singularity_compatibility.html). + +## What is Apptainer? +Apptainer is an open-source container platform designed to be simple, fast, and secure. It allows the portability and reproducibility of operating systems and application environments through the use of Linux containers. It gives users complete control over their environment. + +Apptainer is like Docker but tuned explicitly for HPC clusters. More information is available on the [project’s website](https://apptainer.org/). + +## Features + +- Run the latest applications on an Ubuntu or Centos userland +- Gain access to the latest developer tools +- Launch MPI programs easily +- Much more + +Apptainer’s user guide is available at: [https://apptainer.org/docs/user/main/introduction.html](https://apptainer.org/docs/user/main/introduction.html) + +## Example +Here is an example using an Ubuntu 16.04 image on Negishi: + +```bash +apptainer exec /depot/itap/singularity/ubuntu1604.img cat /etc/lsb-release +DISTRIB_ID=Ubuntu +DISTRIB_RELEASE=16.04 +DISTRIB_CODENAME=xenial +DISTRIB_DESCRIPTION="Ubuntu 16.04 LTS" +``` + +Here is another example using a Centos 7 image: + +```bash +apptainer exec /depot/itap/singularity/centos7.img cat /etc/redhat-release +CentOS Linux release 7.3.1611 (Core) +``` + +## Purdue Cluster Specific Notes + +All service providers will integrate Apptainer slightly differently depending on site. The largest customization will be which default files are inserted into your images so that routine services will work. + +Services we configure for your images include DNS settings and account information. File systems we overlay into your images are your home directory, scratch, Data Depot, and application file systems. + +Here is a list of paths: + +- /etc/resolv.conf +- /etc/hosts +- /home/$USER +- /apps +- /scratch +- /depot + +This means that within the container environment these paths will be present and the same as outside the container. The ```/apps```, ```/scratch```, and ```/depot``` directories will need to exist _inside_ your container to work properly. + +## Creating Apptainer Images +You can build on your system or straight on the cluster (you do not need root privileges to build or run the container). + +You can find information and documentation for how to install and use Apptainer on your system: + +- [Install Apptainer on Windows or MacOS](https://apptainer.org/docs/admin/main/installation.html#installation-on-windows-or-mac) +- [Install Apptainer on Linux](https://apptainer.org/docs/admin/main/installation.html#installation-on-linux) + +We have version ```1.1.6``` (or newer) on the cluster. Please note that installed versions may change throughout cluster life time, so when in doubt, please check exact version with a ```--version``` command line flag: + +```bash +apptainer --version +apptainer version 1.3.3-1.el9 +``` + +Everything you need on how to [build a container](https://apptainer.org/docs/user/main/build_a_container.html) is available from their user guide. Below are merely some quick tips for getting your own containers built for Negishi. + +You can use a [Definition File](https://apptainer.org/docs/user/main/definition_files.html) to both build your container and share its specification with collaborators (for the sake of reproducibility). Here is a simplistic example of such a file: + +```bash +# FILENAME: Buildfile + +Bootstrap: docker +From: ubuntu:18.04 + +%post + apt-get update && apt-get upgrade -y + mkdir /apps /depot /scratch +``` + +To build the image itself: + +```bash +apptainer build ubuntu-18.04.sif Buildfile +``` + +The challenge with this approach however is that it must start from scratch if you decide to change something. In order to create a container image iteratively and interactively, you can use the ```--sandbox``` option. + +```bash +apptainer build --sandbox ubuntu-18.04 docker://ubuntu:18.04 +``` + +This will not create a flat image file but a directory tree (i.e., a folder), the contents of which are the container's filesystem. In order to get a shell inside the container that allows you to modify it, user the ```--writable``` option. + +```bash +apptainer shell --writable ubuntu-18.04 +Apptainer> +``` + +You can then proceed to install any libraries, software, etc. within the container. Then to create the final image file, ```exit``` the shell and call the ```build``` command once more on the sandbox. + +```bash +apptainer build ubuntu-18.04.sif ubuntu-18.04 +``` + +Finally, copy the new image to Negishi and run it. + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/biocontainers.md b/docs/userguides/negishi/run_jobs/biocontainers.md new file mode 100644 index 00000000..e6d23a38 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/biocontainers.md @@ -0,0 +1,87 @@ +--- +tags: + - Negishi +authors: + - jin456 + - remender + - hkashgar +resource: Negishi +search: + boost: 2 +--- +# BioContainers Collection + +What is BioContainers? +---------------------- + +The BioContainers project came from the idea of using the containers-based technologies such as [Docker](https://www.docker.com) or [rkt](https://github.com/rkt/rkt) for bioinformatics software. Having a common and controllable environment for running software could help to deal with some of the current problems during software development and distribution. BioContainers is a community-driven project that provides the infrastructure and basic guidelines to create, manage and distribute bioinformatics containers with a special focus on omics fields such as proteomics, genomics, transcriptomics and metabolomics. . For more information, please visit [BioContainers project](https://biocontainers.pro). + +### Getting Started + +Users can download bioinformatic containers from the [BioContainers.pro](https://biocontainers.pro) and run them directly using Singularity instructions from the corresponding container’s catalog page. + +Brief Singularity guide and examples are available at the [Negishi Singularity user guide](apptainer.md) page. Detailed Singularity user guide is available at: [sylabs.io/guides/3.8/user-guide](https://sylabs.io/guides/3.8/user-guide/) + +In addition, a subset of pre-downloaded biocontainers wrapped into convenient software modules are provided. These modules wrap underlying complexity and provide the same commands that are expected from non-containerized versions of each application. + +On Negishi, type the command below to see the lists of biocontainers we deployed. + +``` +module load biocontainers +module avail + +------------ BioContainers collection modules ------------- + bamtools/2.5.1 + beast2/2.6.3 + bedtools/2.30.0 + blast/2.11.0 + bowtie2/2.4.2 + bwa/0.7.17 + cufflinks/2.2.1 + deeptools/3.5.1 + fastqc/0.11.9 + faststructure/1.0 + htseq/0.13.5 +[....] +``` + +### Example + +This example demonstrates how to run BLASTP with the `blast` module. This `blast` module is a biocontainer wrapper for [NCBI BLAST](https://blast.ncbi.nlm.nih.gov). + +``` +module load biocontainers +module load blast +blastp -query query.fasta -db nr -out output.txt -outfmt 6 -evalue 0.01 +``` + +To run a job in batch mode, first prepare a job script that specifies the BioContainer modules you want to launch and the resources required to run it. Then, use the `sbatch` command to submit your job script to Slurm. The following example shows the job script to use [Bowtie2](http://bowtie-bio.sourceforge.net/bowtie2/index.shtml) in bioinformatic analysis. + +``` +#!/bin/bash + +#SBATCH -A myqueuename +#SBATCH -o bowtie2_%j.txt +#SBATCH -e bowtie2_%j.err +#SBATCH --nodes=1 +#SBATCH --ntasks-per-node=1 +#SBATCH --cpus-per-task=8 +#SBATCH --time=1:30:00 +#SBATCH --job-name bowtie2 + +# Load the Bowtie module +module load biocontainers +module load bowtie2 + +# Indexing a reference genome +bowtie2-build ref.fasta ref + +# Aligning paired-end reads +bowtie2 -p 8 -x ref -1 reads_1.fq -2 reads_2.fq -S align.sam +``` + +To help users get started, we provided detailed user guides for each containerized bioinformatics module on the [ReadTheDocs platform](https://biocontainer-doc.readthedocs.io/en/latest/) + +![RCAC Biocontainers one ReadTheDocs](../../../../assets/images/userguides/examples/biocontainers.png) + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/cancelling_job.md b/docs/userguides/negishi/run_jobs/cancelling_job.md new file mode 100644 index 00000000..30a3e4c7 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/cancelling_job.md @@ -0,0 +1,13 @@ +--- +tags: + - Negishi +authors: + - hkashgar +cluster: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/cancelling_job.md" + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/checking_output.md b/docs/userguides/negishi/run_jobs/checking_output.md new file mode 100644 index 00000000..7439f281 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/checking_output.md @@ -0,0 +1,13 @@ +--- +tags: + - Negishi +authors: + - hkashgar +cluster: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/checking_output.md" + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/creating_the_submission_script.md b/docs/userguides/negishi/run_jobs/creating_the_submission_script.md new file mode 100644 index 00000000..21f2c913 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/creating_the_submission_script.md @@ -0,0 +1,53 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Creatting the Slurm Job Submission Script + +### Script + +To submit work to a SLURM queue, you must first create a job submission file. This job submission file is essentially a simple shell script that includes special comments to specify sbatch options. It will set any required environment variables, load any necessary modules, create or modify files and directories, and run any applications that you need. A simple submission script to the {{ resource }} cpu partition looks like: + +``` bash +#!/bin/bash +# FILENAME: myjobsubmissionfile + +#SBATCH --account=myLabAccount +#SBATCH --partition=cpu +#SBATCH --qos=normal +#SBATCH --nodes=1 +#SBATCH --ntasks=1 +#SBATCH --time=1-00:00:00 + +# Loads Matlab and sets the application up +module load matlab + +# Change to the directory from which you originally submitted this job. +cd $SLURM_SUBMIT_DIR + +# Runs a Matlab script named 'myscript' +matlab -nodisplay -singleCompThread -r myscript +``` +Once your script is prepared, you are ready to [submit your job](./submit_script.md). + +### Job Script Environment Variables + +SLURM sets several potentially useful environment variables which you may use within your job submission files. Here is a list of some: + +| Name | Description | +| --- | --- | +| SLURM\_SUBMIT\_DIR | Absolute path of the current working directory when you submitted this job | +| SLURM\_JOBID | Job ID number assigned to this job by the batch system | +| SLURM\_JOB\_NAME | Job name supplied by the user | +| SLURM\_JOB\_NODELIST | Names of nodes assigned to this job | +| SLURM\_CLUSTER\_NAME | Name of the cluster executing the job | +| SLURM\_SUBMIT\_HOST | Hostname of the system where you submitted this job | +| SLURM\_JOB\_PARTITION | Name of the original queue to which you submitted this job | + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/directives.md b/docs/userguides/negishi/run_jobs/directives.md new file mode 100644 index 00000000..cc15bbd9 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/directives.md @@ -0,0 +1,42 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# SLurm Directives + +So far these examples have shown submitting jobs with the resource requests on the ```sbatch``` command line such as: + +```bash +sbatch -A standby --nodes=1 --time=00:01:00 hello.sub +``` + +The resource requests can also be put into job submission file itself. Documenting the resource requests in the job submission is desirable because the job can be easily reproduced later. Details left in your command history are quickly lost. Arguments are specified with the ```#SBATCH``` syntax: + +```bash +#!/bin/bash + +# FILENAME: hello.sub +#SBATCH -A myallocation -p queue-name +#SBATCH --nodes=1 --time=00:01:00 + +# Show this ran on a compute node by running the hostname command. +hostname + +echo "Hello World" +``` + +The ```#SBATCH``` directives must appear at the top of your submission file. SLURM will stop parsing directives as soon as it encounters a line that does not start with '#'. If you insert a directive in the middle of your script, it will be ignored. + +This job can be then submitted with: + +```bash +sbatch hello.sub +``` + +[**Back to the Example Jobs section**](generic_slurm_jobs.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/gaussian.md b/docs/userguides/negishi/run_jobs/gaussian.md new file mode 100644 index 00000000..30476f2c --- /dev/null +++ b/docs/userguides/negishi/run_jobs/gaussian.md @@ -0,0 +1,131 @@ +--- +tags: + - Negishi +authors: + - jin456 + - remender + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Gaussian + +Gaussian is a computational chemistry software package which works on electronic structure. This section illustrates how to submit a small Gaussian job to a Slurm queue. This Gaussian example runs the Fletcher-Powell multivariable optimization. + +## Gaussian Input File + +Prepare a Gaussian input file with an appropriate filename. In this example, the file is named `myjob.com`. + +The final blank line is necessary. + +```text +#P TEST OPT=FP STO-3G OPTCYC=2 + +STO-3G FLETCHER-POWELL OPTIMIZATION OF WATER + +0 1 +O +H 1 R +H 1 R 2 A +R 0.96 +A 104. +``` + +## Submit the Job + +To submit this job, load Gaussian and then run the provided script, named `subg16`. + +This job uses one compute node with 16 processor cores. + +```bash +module load gaussian16 +subg16 myjob -N 1 -n 16 --gres=gpu:1 +``` + +## View Job Status + +View job status with: + +```bash +squeue -u myusername +``` + +## View Results + +View results in the Gaussian output file. In this example, the output file is named `myjob.log`. + +Only the first and last few lines are shown here: + +```text + Entering Gaussian System, Link 0=/apps/cent7/gaussian/g16-A.03/g16-haswell/g16/g16 + Initial command: + /apps/cent7/gaussian/g16-A.03/g16-haswell/g16/l1.exe /scratch/negishi/myusername/gaussian/Gau-7781.inp -scrdir=/scratch/negishi/myusername/gaussian/ + Entering Link 1 = /apps/cent7/gaussian/g16-A.03/g16-haswell/g16/l1.exe PID= 7782. + + Copyright (c) 1988,1990,1992,1993,1995,1998,2003,2009,2016, + Gaussian, Inc. All Rights Reserved. + +. +. +. + Job cpu time: 0 days 0 hours 3 minutes 28.2 seconds. + Elapsed time: 0 days 0 hours 0 minutes 12.9 seconds. + File lengths (MBytes): RWF= 17 Int= 0 D2E= 0 Chk= 2 Scr= 2 + Normal termination of Gaussian 16 at Tue May 1 17:12:00 2018. +real 13.85 +user 202.05 +sys 6.12 +Machine: +a003.negishi.rcac.purdue.edu +a003.negishi.rcac.purdue.edu +a003.negishi.rcac.purdue.edu +a003.negishi.rcac.purdue.edu +a003.negishi.rcac.purdue.edu +a003.negishi.rcac.purdue.edu +a003.negishi.rcac.purdue.edu +a003.negishi.rcac.purdue.edu +``` + +## Examples of Gaussian SLURM Job Submissions + +Submit a job using 16 processor cores on a single node: + +```bash +subg16 myjob -N 1 -n 16 --gres=gpu:1 -t 24:00:00 -A standby +``` + +Submit a job using 16 processor cores on each of 2 nodes: + +```bash +subg16 myjob -N 2 --ntasks-per-node=16 --gres=gpu:2 -t 24:00:00 -A standby +``` + +## Bash Job Submission Script + +To submit a bash job, a sample submit script looks like this: + +```bash +#!/bin/bash +#SBATCH -A accountname # Queue name; use the 'slist' command to find queue names +#SBATCH --nodes=1 # Total number of nodes +#SBATCH --ntasks=64 # Total number of MPI tasks +#SBATCH --gpus-per-node=1 # Total number of GPUs +#SBATCH --time=1:00:00 # Total run time limit (hh:mm:ss) +#SBATCH -J myjobname # Job name +#SBATCH -o myjob.o%j # Name of stdout output file +#SBATCH -e myjob.e%j # Name of stderr error file +#SBATCH --partition=a10 +#SBATCH --mem=8G + +module load gaussian16 + +g16 < myjob.com +``` + +## Additional Resources + +- [Gaussian Website](https://www.gaussian.com) + +[**Back to the Running Jobs section**](index.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/generic_slurm_jobs.md b/docs/userguides/negishi/run_jobs/generic_slurm_jobs.md new file mode 100644 index 00000000..12da83f6 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/generic_slurm_jobs.md @@ -0,0 +1,26 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +hide: + - toc +--- + +# Generic SLURM Jobs + +The following examples demonstrate the basics of SLURM jobs, and are designed to cover common job request scenarios. These example jobs will need to be modified to run your application or code. + +- [**Simple Job**](simple_job.md) +- [**Multiple Node Job**](multiple_node.md) +- [**Directives**](directives.md) +- [**Specific Types of Nodes**](specific_nodes.md) +- [**Interactive Jobs**](interactive_jobs.md) +- [**Serial Jobs**](serial_jobs.md) +- [**Monitoring Resources**](monitoring_resources.md) + + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/holding_job.md b/docs/userguides/negishi/run_jobs/holding_job.md new file mode 100644 index 00000000..cb1cebda --- /dev/null +++ b/docs/userguides/negishi/run_jobs/holding_job.md @@ -0,0 +1,13 @@ +--- +tags: + - Negishi +authors: + - hkashgar +cluster: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/holding_job.md" + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/index.md b/docs/userguides/negishi/run_jobs/index.md new file mode 100644 index 00000000..a2b6878a --- /dev/null +++ b/docs/userguides/negishi/run_jobs/index.md @@ -0,0 +1,51 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Running Jobs + +Jobs are submitted on {{ resource }} via the SLURM (Simple Linux Utility for Resource Management) scheduler, which is responsible for allocating resources and scheduling the start time of a job. You may use either the batch or interactive mode to run your jobs. The batch mode is ideal for finished programs, and the interactive mode is useful for debugging your job. + +!!! important + Do NOT run large, long, multi-threaded, parallel, or CPU-intensive jobs on a front-end login host. All users share the front-end hosts, and running anything but the smallest test job will negatively impact everyone's ability to use Negishi. Always use SLURM to submit your work as a job. + +Before creating your submission script, learn more about how to use Slurm accounts, partitions, and QOS options: + +- [**Basics of using Slurm accounts, partitions, and QOS options**](queues.md) + - [**Job Submission Matrix**](job_submission_matrix.md) + +Batch jobs submitted via SLURM have four main steps: + +- [**Creating the submission script**](creating_the_submission_script.md) +- [**Submitting the script as a job**](submit_script.md) +- [**Monitoring the job**](monitoring_job.md) +- [**Checking the job output**](checking_output.md) + +### Other useful topics + +- [**Holding a job**](holding_job.md) +- [**Job Dependencies**](job_dependencies.md) +- [**Cancelling a job**](cancelling_job.md) + + +### Example jobs + +- [**Generic SLURM jobs**](generic_slurm_jobs.md) +- [**Python**](python.md) +- [**R**](r.md) +- [**Apptainer**](apptainer.md) +- [**Biocontainers**](biocontainers.md) +- [**Matlab**](matlab.md) +- [**Ansys**](ansysfluent.md) +- [**Gaussian**](gaussian.md) +- [**VASP**](vasp.md) +- [**Windows**](windows.md) +- [**MPI**](mpi_jobs.md) +- [**OpenMP**](openmp_jobs.md) + diff --git a/docs/userguides/negishi/run_jobs/interactive_jobs.md b/docs/userguides/negishi/run_jobs/interactive_jobs.md new file mode 100644 index 00000000..36b9d370 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/interactive_jobs.md @@ -0,0 +1,31 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Running interactive jobs on Negishi + +Interactive jobs are run on compute nodes, while giving you a shell to interact with. They give you the ability to type commands or use a graphical interface in the same way as if you were on a front-end login host. + +To submit an interactive job, use ```sinteractive``` to run a login shell on allocated resources. + +```sinteractive``` accepts most of the same resource requests as ```sbatch```, so to request a login shell on the cpu account while allocating 2 nodes and 128 total cores, you might do: + +```bash +sinteractive -A cpu -N2 -n256 +``` + +To quit your interactive job: + +```bash +exit or Ctrl-D +``` + +The above example will allocate the total of 256 CPU cores across 2 nodes. Note that if your multi-node job requests fewer than each node's full 128 cores per node, by default Slurm provides no guarantee with respect to how this total is distributed between assigned nodes (i.e. the cores may not necessarily be split evenly). If you need specific arrangements of your tasks and cores, you can use ```--cpus-per-task=``` and/or ```--ntasks-per-node=``` flags. See [Slurm documentation](https://slurm.schedmd.com/salloc.html) or ```man salloc``` for more options. + +[**Back to the Example Jobs section**](generic_slurm_jobs.md) diff --git a/docs/userguides/negishi/run_jobs/job_dependencies.md b/docs/userguides/negishi/run_jobs/job_dependencies.md new file mode 100644 index 00000000..62b4d495 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/job_dependencies.md @@ -0,0 +1,13 @@ +--- +tags: + - Negishi +authors: + - hkashgar +cluster: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/job_dependencies.md" + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/job_submission_matrix.md b/docs/userguides/negishi/run_jobs/job_submission_matrix.md new file mode 100644 index 00000000..9510ebb6 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/job_submission_matrix.md @@ -0,0 +1,27 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +hide: + - toc + +--- + +# Job Submission Matrix + + +| Job Type | Partition | QoS | Job Submission Options | Number of Cores Per Account | Number of Jobs Per Account | Priority Accrual | Max Walltime | +| --- | --- | --- | --- | --- | --- | --- | --- | +| PI Queue | cpu | normal | `-A "mygroup" -p cpu` | Limited to purchased cores | No limit | No Limit | 2 weeks | +| Standby Job | cpu | standby | `-A "mygroup" -p cpu -q standby` | 14272 Cores | 5000 | No Limit | 4 hours | +| Highmem Job | highmem | normal | `-A "mygroup" -p highmem` | 128 Cores | 2 | 1 | 24 hours | +| GPU Job | gpu | normal | `-A "mygroup" -p gpu` | 64 cores/3 GPUs/1 Node | 2 | 1 | 24 hours | +| Interactive-Tier Queues | interactive | normal | `-A "mygroup" -p interactive` | 4 cores | 1 per user | 1 | 24 hours | + +Note: The normal QOS is the default and does not need to be specified. + +[**Back to the Running Jobs section**](index.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/matlab.md b/docs/userguides/negishi/run_jobs/matlab.md new file mode 100644 index 00000000..cc1bfb3a --- /dev/null +++ b/docs/userguides/negishi/run_jobs/matlab.md @@ -0,0 +1,33 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Matlab + +*MATLAB®* (MATrix LABoratory) is a high-level language and interactive environment for numerical computation, visualization, and programming. MATLAB is a product of [MathWorks](http://www.mathworks.com/). + +MATLAB, Simulink, Compiler, and several of the optional toolboxes are available to faculty, staff, and students. To see the kind and quantity of all MATLAB licenses plus the number that you are currently using you can use the `matlab_licenses` command: + +```bash +$ module load matlab +$ matlab_licenses +``` + +The MATLAB client can be run in the front-end for application development, however, computationally intensive jobs must be run on compute nodes. + +The following sections provide several examples illustrating how to submit MATLAB jobs to a Linux compute cluster. + +* [**Matlab Script (`.m` File)**](./matlab/interpreter.md) +* [**Implicit Parallelism**](./matlab/implicit_parallelism.md) +* [**Profile Manager**](./matlab/profile_manager.md) +* [**Parallel Computing Toolbox (parfor)**](./matlab/parfor.md) +* [**Parallel Toolbox (spmd)**](./matlab/spmd.md) +* [**Distributed Computing Server (parallel job)**](./matlab/mdcs_parallel.md) + +[**Back to the Running Jobs section**](index.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/matlab/implicit_parallelism.md b/docs/userguides/negishi/run_jobs/matlab/implicit_parallelism.md new file mode 100644 index 00000000..5b6b6e0d --- /dev/null +++ b/docs/userguides/negishi/run_jobs/matlab/implicit_parallelism.md @@ -0,0 +1,35 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Implicit Parallelism + +MATLAB implements *implicit parallelism* which is automatic multithreading of many computations, such as matrix multiplication, linear algebra, and performing the same operation on a set of numbers. This is different from the explicit parallelism of the Parallel Computing Toolbox. + +MATLAB offers implicit parallelism in the form of thread-parallel enabled functions. Since these processor cores, or threads, share a common memory, many MATLAB functions contain multithreading potential. Vector operations, the particular application or algorithm, and the amount of computation (array size) contribute to the determination of whether a function runs serially or with multithreading. + +When your job triggers implicit parallelism, it attempts to allocate its threads on all processor cores of the compute node on which the MATLAB client is running, including processor cores running other jobs. This competition can degrade the performance of all jobs running on the node. + +!!! note + When you know that you are coding a serial job but are unsure whether you are using thread-parallel enabled operations, run MATLAB with implicit parallelism turned off. Beginning with the R2009b, you can turn multithreading off by starting MATLAB with `-singleCompThread`: + + ```bash + $ matlab -nodisplay -singleCompThread -r mymatlabprogram + ``` + +When you are using implicit parallelism, make sure you request exclusive access to a compute node, as MATLAB has no facility for sharing nodes. + +For more information about MATLAB's implicit parallelism: + +* [Which MATLAB functions benefit from multithreaded computation?](http://www.mathworks.com/support/solutions/en/data/1-4PG4AN/index.html?solution=1-4PG4AN) +* [What is the difference between "MATLAB as a fully-multithreaded application" versus "multithreaded computation"?](http://www.mathworks.com/support/solutions/en/data/1-3P8CC5/index.html) +* [MathWorks Website](http://www.mathworks.com/) + + +[**Back to Matlab**](../matlab.md) diff --git a/docs/userguides/negishi/run_jobs/matlab/interpreter.md b/docs/userguides/negishi/run_jobs/matlab/interpreter.md new file mode 100644 index 00000000..92551876 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/matlab/interpreter.md @@ -0,0 +1,98 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Matlab Script (.m File) + +This section illustrates how to submit a small, serial, MATLAB program as a job to a batch queue. This MATLAB program prints the name of the run host and gets three random numbers. + +Prepare a MATLAB script `myscript.m`, and a MATLAB function file `myfunction.m`: + +```M +% FILENAME: myscript.m + +% Display name of compute node which ran this job. +[c name] = system('hostname'); +fprintf('\n\nhostname:%s\n', name); + +% Display three random numbers. +A = rand(1,3); +fprintf('%f %f %f\n', A); + +quit; +``` + +```M +% FILENAME: myfunction.m + +function result = myfunction () + + % Return name of compute node which ran this job. + [c name] = system('hostname'); + result = sprintf('hostname:%s', name); + + % Return three random numbers. + A = rand(1,3); + r = sprintf('%f %f %f', A); + result=strvcat(result,r); + +end +``` + +Also, prepare a job submission file, here named `myjob.sub`. Run with the name of the script: + +```bash +#!/bin/bash +# FILENAME: myjob.sub + +echo "myjob.sub" + +# Load module, and set up environment for Matlab to run +module load matlab + +unset DISPLAY + +# -nodisplay: run MATLAB in text mode; X11 server not needed +# -singleCompThread: turn off implicit parallelism +# -r: read MATLAB program; use MATLAB JIT Accelerator +# Run Matlab, with the above options and specifying our .m file +matlab -nodisplay -singleCompThread -r myscript +``` + +[Submit the job](../submit_script.md) + +[View job status](../monitoring_job.md) + +[View results of the job](../checking_output.md) + +```bash +myjob.sub + + < M A T L A B (R) > + Copyright 1984-2011 The MathWorks, Inc. + R2011b (7.13.0.564) 64-bit (glnxa64) + August 13, 2011 + +To get started, type one of these: helpwin, helpdesk, or demo. +For product information, visit www.mathworks.com. + +hostname: a001.negishi.rcac.purdue.edu +0.814724 0.905792 0.126987 +``` + +Output shows that a processor core on one compute node (a001) processed the job. Output also displays the three random numbers. + +For more information about MATLAB: + +* [inv()](http://www.mathworks.com/help/techdoc/ref/inv.html) +* [Run a Batch Job](http://www.mathworks.com/help/distcomp/introduction-to-parallel-solutions.html#brjw1fx-2) +* [Archived MathWorks Documentation](http://www.mathworks.com/help/doc-archives.html) +* [MathWorks Website](http://www.mathworks.com/) + +[**Back to Matlab**](../matlab.md) diff --git a/docs/userguides/negishi/run_jobs/matlab/mdcs_parallel.md b/docs/userguides/negishi/run_jobs/matlab/mdcs_parallel.md new file mode 100644 index 00000000..934311c3 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/matlab/mdcs_parallel.md @@ -0,0 +1,164 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Distributed Computing Server (parallel job) + +The MATLAB Parallel Computing Toolbox (PCT) enables a parallel job via the MATLAB Distributed Computing Server (DCS). The tasks of a parallel job are identical, run simultaneously on several MATLAB workers, or labs, and communicate with each other. + +This example illustrates an MPI-like program. + +The example shows how to submit a small MATLAB parallel job with four workers running one MPI-like task to a batch queue. The MATLAB program broadcasts an integer to four workers and gathers the names of the compute nodes running the workers and the lab IDs of the workers. + +This example uses the job submission command to submit a MATLAB script with a user-defined cluster profile. The profile scatters the MATLAB workers onto different compute nodes. + +This method uses: + +- the MATLAB interpreter +- the Parallel Computing Toolbox +- the Distributed Computing Server + +It requires and checks out six licenses: + +- one MATLAB license for the client running on the compute node +- one Parallel Computing Toolbox license +- four Distributed Computing Server licenses + +Four DCS licenses run the four copies of the parallel job. + +This job is completely off the front end. + +## MATLAB Script + +Prepare a MATLAB script named `myscript.m`: + +```matlab +% FILENAME: myscript.m + +% Specify pool size. +% Convert the parallel job to a pool job. +parpool('4'); +spmd + +if labindex == 1 + % Lab (rank) #1 broadcasts an integer value to other labs (ranks). + N = labBroadcast(1,int64(1000)); +else + % Each lab (rank) receives the broadcast value from lab (rank) #1. + N = labBroadcast(1); +end + +% Form a string with host name, total number of labs, lab ID, and broadcast value. +[c name] =system('hostname'); +name = name(1:length(name)-1); +fmt = num2str(floor(log10(numlabs))+1); +str = sprintf(['%s:%d:%' fmt 'd:%d '], name,numlabs,labindex,N); + +% Apply global concatenate to all str's. +% Store the concatenation of str's in the first dimension (row) and on lab #1. +result = gcat(str,1,1); +if labindex == 1 + disp(result) +end + +end % spmd +matlabpool close force; +quit; +``` + +## SLURM Job Submission File + +Prepare a job submission file. In this example, the file is named `myjob.sub`. + +Run with the name of the script: + +```bash +# FILENAME: myjob.sub + +echo "myjob.sub" + +module load matlab + +unset DISPLAY + +# -nodisplay: run MATLAB in text mode; X11 server not needed +# -r: read MATLAB program; use MATLAB JIT Accelerator +matlab -nodisplay -r myscript +``` + +## Set the Default Parallel Configuration + +Run MATLAB to set the default parallel configuration to your appropriate profile: + +```bash +matlab -nodisplay +``` + +Then, inside MATLAB: + +```matlab +defaultParallelConfig('myslurmprofile'); +quit; +``` + +## Submit the Job + +Submit the job as a single compute node with one processor core. + +Once this job starts, a second job submission is made. + +## Example Output + +The output may look similar to this: + +```text +myjob.sub + < M A T L A B (R) > + Copyright 1984-2011 The MathWorks, Inc. + R2011b (7.13.0.564) 64-bit (glnxa64) + August 13, 2011 + +To get started, type one of these: helpwin, helpdesk, or demo. +For product information, visit www.mathworks.com. +>Starting matlabpool using the 'myslurmprofile' configuration ... connected to 4 labs. +Lab 1: + negishi.a006.rcac.purdue.edu:4:1:1000 + negishi.a007.rcac.purdue.edu:4:2:1000 + negishi.a008.rcac.purdue.edu:4:3:1000 + negishi.a009.rcac.purdue.edu:4:4:1000 +Sending a stop signal to all the labs ... stopped. +Did not find any pre-existing parallel jobs created by matlabpool. +``` + +The output shows the name of one compute node, `a006`, that processed the job submission file `myjob.sub`. + +The job submission scattered four processor cores, or four MATLAB labs, among four different compute nodes: + +- `a006` +- `a007` +- `a008` +- `a009` + +These nodes processed the four parallel regions. + +## Scaling Up + +To scale this method for a real application: + +1. Increase the wall time in the submission command to accommodate a longer-running job. +2. Increase the wall time of `myslurmprofile` by using the MATLAB Cluster Profile Manager. +3. In the Cluster Profile Manager, use the `Parallel` menu to enter a new wall time in the `SubmitArguments` property. + +## Additional Resources + +- [MathWorks MATLAB Parallel Computing Toolbox User's Guide](https://www.mathworks.com/help/parallel-computing/) +- [MathWorks MATLAB Distributed Computing Server User's Guide](https://www.mathworks.com/help/matlab-parallel-server/) +- [MathWorks Website](https://www.mathworks.com/) + +[**Back to Matlab**](../matlab.md) diff --git a/docs/userguides/negishi/run_jobs/matlab/parfor.md b/docs/userguides/negishi/run_jobs/matlab/parfor.md new file mode 100644 index 00000000..fc0b7d13 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/matlab/parfor.md @@ -0,0 +1,145 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Parallel Computing Toolbox (`parfor`) + +The MATLAB Parallel Computing Toolbox (PCT) extends MATLAB with high-level parallel-processing features such as parallel `for` loops, parallel regions, message passing, distributed arrays, and parallel numerical methods. + +This example illustrates the fine-grained parallelism of a parallel `for` loop, or `parfor`, in a pool job. + +The examples below show a method for submitting a small parallel MATLAB program with a `parfor` loop as a job to a queue. The MATLAB program prints the name of the run host and shows the values of the variables `numlabs` and `labindex` for each iteration of the `parfor` loop. + +This method uses the job submission command to submit a MATLAB client, which calls the MATLAB `batch()` function with a user-defined cluster profile. + +## MATLAB `parfor` Script + +Prepare a MATLAB pool program in a MATLAB script with an appropriate filename. In this example, the file is named `myscript.m`. + +```M +% FILENAME: myscript.m +% SERIAL REGION +[c name] = system('hostname'); +fprintf('SERIAL REGION: hostname:%s\n', name) +numlabs = parpool('poolsize'); +fprintf(' hostname numlabs labindex iteration\n') +fprintf(' ------------------------------- ------- -------- ---------\n') +tic; + +% PARALLEL LOOP +parfor i = 1:8 + [c name] = system('hostname'); + name = name(1:length(name)-1); + fprintf('PARALLEL LOOP: %-31s %7d %8d %9d\n', name,numlabs,labindex,i) + pause(2); +end + +% SERIAL REGION +elapsed_time = toc; % get elapsed time in parallel loop +fprintf('\n') +[c name] = system('hostname'); +name = name(1:length(name)-1); +fprintf('SERIAL REGION: hostname:%s\n', name) +fprintf('Elapsed time in parallel loop: %f\n', elapsed_time) +``` + +The execution of a pool job starts with a worker executing the statements of the first serial region up to the `parfor` block, where it pauses. A set of workers, called the pool, executes the `parfor` block. When they finish, the first worker resumes by executing the second serial region. + +The code displays the names of the compute nodes running the batch session and the worker pool. + +## MATLAB Batch Script + +Prepare a MATLAB script that calls the MATLAB `batch()` function. This creates a four-lab pool on which to run the MATLAB code in `myscript.m`. + +In this example, the file is named `mylclbatch.m`. + +```M +% FILENAME: mylclbatch.m + +!echo "mylclbatch.m" +!hostname + +pjob=batch('myscript','Profile','myslurmprofile','Pool',4,'CaptureDiary',true); +wait(pjob); +diary(pjob); +quit; +``` + +## SLURM Job Submission File + +Prepare a job submission file with an appropriate filename. In this example, the file is named `myjob.sub`. + +```bash +#!/bin/bash +# FILENAME: myjob.sub + +echo "myjob.sub" +hostname + +module load matlab + +unset DISPLAY + +matlab -nodisplay -r mylclbatch +``` + +## Submit the Job + +Submit the job as a single compute node with one processor core. + +One processor core runs `myjob.sub` and `mylclbatch.m`. + +Once this job starts, a second job submission is made by MATLAB through the configured SLURM cluster profile. + +## Example Output + +The output may look similar to this: + +```text +myjob.sub + < M A T L A B (R) > + Copyright 1984-2013 The MathWorks, Inc. + R2013a (8.1.0.604) 64-bit (glnxa64) + February 15, 2013 + +To get started, type one of these: helpwin, helpdesk, or demo. +For product information, visit www.mathworks.com. + +mylclbatch.m a000.negishi.rcac.purdue.edu +SERIAL REGION: hostname:a000.negishi.rcac.purdue.edu + hostname numlabs labindex iteration + ------------------------------- ------- -------- --------- +PARALLEL LOOP: a001.negishi.rcac.purdue.edu 4 1 2 +PARALLEL LOOP: a002.negishi.rcac.purdue.edu 4 1 4 +PARALLEL LOOP: a001.negishi.rcac.purdue.edu 4 1 5 +PARALLEL LOOP: a002.negishi.rcac.purdue.edu 4 1 6 +PARALLEL LOOP: a003.negishi.rcac.purdue.edu 4 1 1 +PARALLEL LOOP: a003.negishi.rcac.purdue.edu 4 1 3 +PARALLEL LOOP: a004.negishi.rcac.purdue.edu 4 1 7 +PARALLEL LOOP: a004.negishi.rcac.purdue.edu 4 1 8 +SERIAL REGION: hostname:a001.negishi.rcac.purdue.edu + +Elapsed time in parallel loop: 5.411486 +``` + +## Scaling Up + +To scale this method for a real application: + +1. Increase the wall time in the SLURM submission command to accommodate a longer-running job. +2. Increase the wall time of `myslurmprofile` by using the MATLAB Cluster Profile Manager. +3. In the Cluster Profile Manager, use the `Parallel` menu to enter a new wall time in the `SubmitArguments` property. + +## Additional Resources + +- [MathWorks MATLAB Parallel Computing Toolbox User's Guide](https://www.mathworks.com/help/parallel-computing/) +- [MathWorks MATLAB Parallel Server Documentation](https://www.mathworks.com/help/matlab-parallel-server/) +- [MathWorks Website](https://www.mathworks.com/) + +[**Back to Matlab**](../matlab.md) diff --git a/docs/userguides/negishi/run_jobs/matlab/profile_manager.md b/docs/userguides/negishi/run_jobs/matlab/profile_manager.md new file mode 100644 index 00000000..806f6981 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/matlab/profile_manager.md @@ -0,0 +1,29 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Profile Manager + +MATLAB offers two kinds of profiles for parallel execution: the `local` profile and user-defined cluster profiles. The 'local' profile runs a MATLAB job on the processor core(s) of the same compute node, or front-end, that is running the client. To run a MATLAB job on compute node(s) different from the node running the client, you must define a Cluster Profile using the `Cluster Profile Manager`. + +To prepare a user-defined cluster profile, use the `Cluster Profile Manager` in the `Parallel` menu. This profile contains the scheduler details (queue, nodes, processors, walltime, etc.) of your job submission. Ultimately, your cluster profile will be an argument to MATLAB functions like `batch()`. + +For your convenience, a generic cluster profile is provided that can be downloaded: [`myslurmprofile.settings`](../../../../assets/scripts/userguides/myslurmprofile.settings) + +Please note that modifications are very likely to be required to make `myslurmprofile.settings` work. You may need to change values for number of nodes, number of workers, walltime, and submission queue specified in the file. As well, the generic profile itself depends on the particular job scheduler on the cluster, so you may need to download or create two or more generic profiles under different names. Each time you run a job using a Cluster Profile, make sure the specific profile you are using is appropriate for the job and the cluster. + +To import the profile, start a MATLAB session and select `Manage Cluster Profiles...` from the Parallel menu. In the Cluster Profile Manager, select `Import`, navigate to the folder containing the profile, select `myslurmprofile.settings` and click `OK`. Remember that the profile will need to be customized for your specific needs. If you have any questions, please contact us. + +For detailed information about MATLAB's Parallel Computing Toolbox, examples, demos, and tutorials: + +* [MATLAB - Parallel Computing Toolbox](http://www.mathworks.com/help/distcomp/index.html) +* [MATLAB Parallel Computing Toolbox: Introduction to Parallel Solutions](http://www.mathworks.com/help/distcomp/introduction-to-parallel-solutions.html) +* [MATLAB Parallel Computing Toolbox: Clusters and Cluster Profiles](https://www.mathworks.com/help/parallel-computing/discover-clusters-and-use-cluster-profiles.html) + +[**Back to Matlab**](../matlab.md) diff --git a/docs/userguides/negishi/run_jobs/matlab/spmd.md b/docs/userguides/negishi/run_jobs/matlab/spmd.md new file mode 100644 index 00000000..3904b563 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/matlab/spmd.md @@ -0,0 +1,152 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Parallel Toolbox (`spmd`) + +The MATLAB Parallel Computing Toolbox (PCT) extends the MATLAB language with high-level parallel-processing features such as parallel `for` loops, parallel regions, message passing, distributed arrays, and parallel numerical methods. + +It offers a shared-memory computing environment with a maximum number of MATLAB workers running on the local configuration in addition to your MATLAB client. MATLAB Distributed Computing Server (DCS) can scale PCT applications up to the limit of your DCS licenses. + +This example shows how to submit a small parallel MATLAB program with a parallel region, using an `spmd` statement, as a MATLAB pool job to a batch queue. + +This method uses the job submission command to submit a MATLAB client to compute nodes. The MATLAB client interprets a MATLAB `.m` file with a user-defined cluster profile, which scatters the MATLAB workers onto different compute nodes. + +This method uses the MATLAB interpreter, the Parallel Computing Toolbox, and the Distributed Computing Server. It requires and checks out six licenses: + +- one MATLAB license for the client running on the compute node +- one Parallel Computing Toolbox license +- four Distributed Computing Server licenses + +Four DCS licenses run the four copies of the `spmd` statement. This job is completely off the front end. + +## MATLAB `spmd` Script + +Prepare a MATLAB script called `myscript.m`: + +```matlab +% FILENAME: myscript.m + +% SERIAL REGION +[c name] = system('hostname'); +fprintf('SERIAL REGION: hostname:%s\n', name) +p = parpool('4'); +fprintf(' hostname numlabs labindex\n') +fprintf(' ------------------------------- ------- --------\n') +tic; + +% PARALLEL REGION +spmd + [c name] = system('hostname'); + name = name(1:length(name)-1); + fprintf('PARALLEL REGION: %-31s %7d %8d\n', name,numlabs,labindex) + pause(2); +end + +% SERIAL REGION +elapsed_time = toc; % get elapsed time in parallel region +delete(p); +fprintf('\n') +[c name] = system('hostname'); +name = name(1:length(name)-1); +fprintf('SERIAL REGION: hostname:%s\n', name) +fprintf('Elapsed time in parallel region: %f\n', elapsed_time) +quit; +``` + +## SLURM Job Submission File + +Prepare a job submission file with an appropriate filename. In this example, the file is named `myjob.sub`. + +Run with the name of the script: + +```bash +#!/bin/bash +# FILENAME: myjob.sub + +echo "myjob.sub" +module load matlab + +unset DISPLAY + +matlab -nodisplay -r myscript +``` + +## Set the Default Parallel Configuration + +Run MATLAB to set the default parallel configuration to your job configuration: + +```bash +matlab -nodisplay +``` + +Then, inside MATLAB: + +```matlab +parallel.defaultClusterProfile('myslurmprofile'); +quit; +``` + +## Submit the Job + +Submit the job using `sbatch`. + +Once this job starts, a second job submission is made by MATLAB through the configured SLURM cluster profile. + +## Example Output + +The output may look similar to this: + +```text +myjob.sub + < M A T L A B (R) > + Copyright 1984-2011 The MathWorks, Inc. + R2011b (7.13.0.564) 64-bit (glnxa64) + August 13, 2011 + +To get started, type one of these: helpwin, helpdesk, or demo. +For product information, visit www.mathworks.com. + +SERIAL REGION: hostname:negishi.a001.rcac.purdue.edu +Starting matlabpool using the 'myslurmprofile' profile ... connected to 4 labs. + hostname numlabs labindex + ------------------------------- ------- -------- +Lab 2: + PARALLEL REGION: negishi.a002.rcac.purdue.edu 4 2 +Lab 1: + PARALLEL REGION: negishi.a001.rcac.purdue.edu 4 1 +Lab 3: + PARALLEL REGION: negishi.a003.rcac.purdue.edu 4 3 +Lab 4: + PARALLEL REGION: negishi.a004.rcac.purdue.edu 4 4 +Sending a stop signal to all the labs ... stopped. + +SERIAL REGION: hostname:negishi.a001.rcac.purdue.edu +Elapsed time in parallel region: 3.382151 +``` + +The output shows that one compute node, `a001`, processed the job submission file `myjob.sub` and the two serial regions. + +The job submission scattered four processor cores, or four MATLAB labs, among four different compute nodes: + +- `a001` +- `a002` +- `a003` +- `a004` + +These nodes processed the four parallel regions. The total elapsed time demonstrates that the jobs ran in parallel. + +## Additional Resources + +- [MathWorks MATLAB Parallel Computing Toolbox User's Guide](https://www.mathworks.com/help/parallel-computing/) +- [MathWorks MATLAB Distributed Computing Server User's Guide](https://www.mathworks.com/help/matlab-parallel-server/) +- [MathWorks Website](https://www.mathworks.com/) + + +[**Back to Matlab**](../matlab.md) diff --git a/docs/userguides/negishi/run_jobs/monitoring_job.md b/docs/userguides/negishi/run_jobs/monitoring_job.md new file mode 100644 index 00000000..8190bb01 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/monitoring_job.md @@ -0,0 +1,76 @@ +--- +tags: + - Negishi +authors: + - hkashgar +cluster: Negishi +search: + boost: 2 +--- + +# Monitoring the Job + +Once a job is [submitted](./submit_script.md) there are several commands you can use to monitor the progress of the job. + +To see your jobs, use the `squeue -u` command and specify your username: + +(Remember, in our SLURM environment a queue is referred to as an 'Account') + +```bash +squeue -u myusername + + JOBID ACCOUNT NAME USER ST TIME NODES NODELIST(REASON) + 182792 standby job1 myusername R 20:19 1 negishi-a000 + 185841 standby job2 myusername R 20:19 1 negishi-a001 + 185844 standby job3 myusername R 20:18 1 negishi-a002 + 185847 standby job4 myusername R 20:18 1 negishi-a003 +``` + +To retrieve useful information about your queued or running job, use the `scontrol show job` command with your job's ID number. The output should look similar to the following: + +```bash +scontrol show job 3519 + +JobId=3519 JobName=t.sub + UserId=${user.username} GroupId=mygroup MCS_label=N/A + Priority=3 Nice=0 Account=(null) QOS=(null) + JobState=PENDING Reason=BeginTime Dependency=(null) + Requeue=1 Restarts=0 BatchFlag=1 Reboot=0 ExitCode=0:0 + RunTime=00:00:00 TimeLimit=7-00:00:00 TimeMin=N/A + SubmitTime=2019-08-29T16:56:52 EligibleTime=2019-08-29T23:30:00 + AccrueTime=Unknown + StartTime=2019-08-29T23:30:00 EndTime=2019-09-05T23:30:00 Deadline=N/A + PreemptTime=None SuspendTime=None SecsPreSuspend=0 + LastSchedEval=2019-08-29T16:56:52 + Partition=workq AllocNode:Sid=mack-fe00:54476 + ReqNodeList=(null) ExcNodeList=(null) + NodeList=(null) + NumNodes=1 NumCPUs=2 NumTasks=2 CPUs/Task=1 ReqB:S:C:T=0:0:*:* + TRES=cpu=2,node=1,billing=2 + Socks/Node=* NtasksPerN:B:S:C=0:0:*:* CoreSpec=* + MinCPUsNode=1 MinMemoryNode=0 MinTmpDiskNode=0 + Features=(null) DelayBoot=00:00:00 + OverSubscribe=OK Contiguous=0 Licenses=(null) Network=(null) + Command=/home/${user.username}/jobdir/myjobfile.sub + WorkDir=/home/${user.username}/jobdir + StdErr=/home/${user.username}/jobdir/slurm-3519.out + StdIn=/dev/null + StdOut=/home/${user.username}/jobdir/slurm-3519.out + Power= + +``` + +There are several useful bits of information in this output. + +- ```JobState``` lets you know if the job is Pending, Running, Completed, or Held. + +- ```RunTime``` and ```TimeLimit``` will show how long the job has run and its maximum time. + +- ```SubmitTime``` is when the job was submitted to the cluster. + +- ```NumNodes```, ```NumCPUs```, ```NumTasks``` and ```CPUs/Task``` are the number of +Nodes, CPUs, Tasks, and - CPUs per Task are shown. + +- ```WorkDir``` is the job's working directory. + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/monitoring_resources.md b/docs/userguides/negishi/run_jobs/monitoring_resources.md new file mode 100644 index 00000000..cde9ba6c --- /dev/null +++ b/docs/userguides/negishi/run_jobs/monitoring_resources.md @@ -0,0 +1,91 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Monitoring Resources + +## Collecting System Resource Utilization Data + +Knowing the precise resource utilization an application had during a job, such as CPU load or memory, can be incredibly useful. This is especially the case when the application isn't performing as expected. + +One approach is to run a program like ```htop``` during an interactive job and keep an eye on system resources. You can get precise time-series data from nodes associated with your job using [XDmod](https://xdmod.rcac.purdue.edu/) as well, online. But these methods don't gather telemetry in an automated fashion, nor do they give you control over the resolution or format of the data. + +As a matter of course, a robust implementation of some HPC workload would include resource utilization data as a diagnostic tool in the event of some failure. + +The ```monitor``` utility is a simple command line system resource monitoring tool for gathering such telemetry and is available as a module. + +```bash +module load monitor +``` + +Complete documentation is available online at [resource-monitor.readthedocs.io](https://resource-monitor.readthedocs.io/en/latest/). A full manual page is also available for reference, ``man monitor```. + +In the context of a SLURM job you will need to put this monitoring task in the background to allow the rest of your job script to proceed. Be sure to interrupt these tasks at the end of your job. + +```bash +#!/bin/bash +# FILENAME: monitored_job.sh + + module load monitor + +# track per-code CPU load +monitor cpu percent --all-cores >cpu-percent.log & +CPU_PID=$! + +# track memory usage +monitor cpu memory >cpu-memory.log & +MEM_PID=$! + +# your code here + +# shut down the resource monitors +kill -s INT $CPU_PID $MEM_PID +``` + +A particularly elegant solution would be to include such tools in your ```prologue``` script and have the tear down in your ```epilogue``` script. + +For large distributed jobs spread across multiple nodes, ```mpiexec``` can be used to gather telemetry from all nodes in the job. The hostname is included in each line of output so that data can be grouped as such. A concise way of constructing the needed list of hostnames in SLURM is to simply use ```srun hostname | sort -u```. + +```bash +#!/bin/bash +# FILENAME: monitored_job.sh + +module load monitor + +# track all CPUs (one monitor per host) +mpiexec -machinefile <(srun hostname | sort -u) \ + monitor cpu percent --all-cores >cpu-percent.log & +CPU_PID=$! + +# track memory on all hosts (one monitor per host) +mpiexec -machinefile <(srun hostname | sort -u) \ + monitor cpu memory >cpu-memory.log & +MEM_PID=$! + +# your code here + +# shut down the resource monitors +kill -s INT $CPU_PID $MEM_PID +``` + +To get resource data in a more readily computable format, the ```monitor``` program can be told to output in CSV format with the ```--csv``` flag. + +```bash +monitor cpu memory --csv >cpu-memory.csv +``` + +For a distributed job you will need to suppress the header lines otherwise one will be created by each host. + +```bash +monitor cpu memory --csv | head -1 >cpu-memory.csv +mpiexec -machinefile <(srun hostname | sort -u) \ + monitor cpu memory --csv --no-header >>cpu-memory.csv +``` + +[**Back to the Example Jobs section**](generic_slurm_jobs.md) diff --git a/docs/userguides/negishi/run_jobs/mpi_jobs.md b/docs/userguides/negishi/run_jobs/mpi_jobs.md new file mode 100644 index 00000000..14bfaeb8 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/mpi_jobs.md @@ -0,0 +1,121 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# MPI + +An MPI job is a set of processes that take advantage of multiple compute nodes by communicating with each other. OpenMPI and Intel MPI (IMPI) are implementations of the MPI standard. + +This section shows how to submit one of the MPI programs compiled in the section [Compiling MPI Programs](../../compile/mpi). + +Use `module load` to set up the paths to access these libraries. Use `module avail` to see all MPI packages installed on Negishi. + +## Example MPI Job Submission File + +Create a job submission file named `mpi_hello.sub`: + +```bash +#!/bin/bash +# FILENAME: mpi_hello.sub +#SBATCH --nodes=2 +#SBATCH --ntasks-per-node=128 +#SBATCH --time=00:01:00 +#SBATCH -A standby + +srun -n 256 ./mpi_hello +``` + +SLURM can run an MPI program with the `srun` command. The number of processes is requested with the `-n` option. If you do not specify the `-n` option, it will default to the total number of processor cores you request from SLURM. + +If the code is built with OpenMPI, it can be run with a simple `srun -n` command. If it is built with Intel IMPI, then you also need to add the `--mpi=pmi2` option: `srun --mpi=pmi2 -n 256 ./mpi_hello` in this example. + +## Submit the MPI Job + +Submit the MPI job: + +```bash +sbatch ./mpi_hello.sub +``` + +## View Results + +View results in the output file: + +```bash +cat slurm-myjobid.out +``` + +Example output: + +```text +Runhost:a010.negishi.rcac.purdue.edu Rank:0 of 256 ranks hello, world +Runhost:a010.negishi.rcac.purdue.edu Rank:1 of 256 ranks hello, world +... +Runhost:a011.negishi.rcac.purdue.edu Rank:128 of 256 ranks hello, world +Runhost:a011.negishi.rcac.purdue.edu Rank:129 of 256 ranks hello, world +... +``` + +If the job failed to run, view error messages in the output file. + +## Reducing MPI Ranks Per Node for Memory-Heavy Jobs + +If an MPI job uses a lot of memory and 128 MPI ranks per compute node use all of the memory of the compute nodes, request more compute nodes while keeping the total number of MPI ranks unchanged. + +Submit the job with double the number of compute nodes and modify the resource request to halve the number of MPI ranks per compute node. + +Create or modify `mpi_hello.sub`: + +```bash +#!/bin/bash +# FILENAME: mpi_hello.sub +#SBATCH --nodes=4 +#SBATCH --ntasks-per-node=64 +#SBATCH -t 00:01:00 +#SBATCH -A standby + +srun -n 256 ./mpi_hello +``` + +Submit the job: + +```bash +sbatch ./mpi_hello.sub +``` + +View results in the output file: + +```bash +cat slurm-myjobid.out +``` + +Example output: + +```text +Runhost:a010.negishi.rcac.purdue.edu Rank:0 of 256 ranks hello, world +Runhost:a010.negishi.rcac.purdue.edu Rank:1 of 256 ranks hello, world +... +Runhost:a011.negishi.rcac.purdue.edu Rank:64 of 256 ranks hello, world +... +Runhost:a012.negishi.rcac.purdue.edu Rank:128 of 256 ranks hello, world +... +Runhost:a013.negishi.rcac.purdue.edu Rank:192 of 256 ranks hello, world +... +``` + +## Notes + +- Use `slist` to determine which queues, specified by the `--account` or `-A` option, are available to you. +- The queue available to everyone on Negishi is `standby`. +- Invoking an MPI program on Negishi with `./program` is typically wrong, since this will use only one MPI process and defeat the purpose of using MPI. +- Unless using only one MPI process is what you want, which is rarely the case, use `srun` or `mpiexec` to invoke an MPI program. +- In general, the exact order in which MPI ranks write similar output to an output file is random. + + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/multiple_node.md b/docs/userguides/negishi/run_jobs/multiple_node.md new file mode 100644 index 00000000..7d2d6f85 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/multiple_node.md @@ -0,0 +1,34 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Multiple Node Job + +In some cases, you may want to request multiple nodes. To utilize multiple nodes, you will need to have a program or code that is specifically programmed to use multiple nodes such as with MPI. Simply requesting more nodes will not make your work go faster. Your code must support this ability. + +This example shows a request for multiple compute nodes. The job submission file contains a single command to show the names of the compute nodes allocated: + +```bash +# FILENAME: myjobsubmissionfile.sub +#!/bin/bash +echo "$SLURM_JOB_NODELIST" +``` +``` +sbatch --nodes=2 --ntasks=256 --time=00:10:00 -A standby myjobsubmissionfile.sub +``` + +Compute nodes allocated: + +```bash +[014-015].negishi +``` + +The above example will allocate the total of 256 CPU cores across 2 nodes. Note that if your multi-node job requests fewer than each node's full 128 cores per node, by default Slurm provides no guarantee with respect to how this total is distributed between assigned nodes (i.e. the cores may not necessarily be split evenly). If you need specific arrangements of your tasks and cores, you can use `--cpus-per-task=` and/or `--ntasks-per-node=` flags. See [Slurm documentation](https://slurm.schedmd.com/sbatch.html) or `man sbatch` for more options. + +[**Back to the Example Jobs section**](generic_slurm_jobs.md) diff --git a/docs/userguides/negishi/run_jobs/openmp_jobs.md b/docs/userguides/negishi/run_jobs/openmp_jobs.md new file mode 100644 index 00000000..741edbe0 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/openmp_jobs.md @@ -0,0 +1,82 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# OpenMP + +A shared-memory job is a single process that takes advantage of a multi-core processor and its shared memory to achieve parallelization. + +This example shows how to submit an OpenMP program compiled in the section [Compiling OpenMP Programs](../compile/openmp). + +!!! note + When running OpenMP programs, all threads must be on the same compute node to take advantage of shared memory. The threads cannot communicate between nodes. + +## Set `OMP_NUM_THREADS` + +To run an OpenMP program, set the environment variable `OMP_NUM_THREADS` to the desired number of threads. + +In `csh`: + +```csh +setenv OMP_NUM_THREADS 128 +``` + +In `bash`: + +```bash +export OMP_NUM_THREADS=128 +``` + +This should almost always be equal to the number of cores on a compute node. You may want to set it to another appropriate value if you are running several processes in parallel in a single job or node. + +## Example Job Submission File + +Create a job submission file named `omp_hello.sub`: + +```bash +#!/bin/bash +# FILENAME: omp_hello.sub +#SBATCH --nodes=1 +#SBATCH --ntasks=128 +#SBATCH --time=00:01:00 + +export OMP_NUM_THREADS=128 +./omp_hello +``` + +## Submit the Job + +Submit the job: + +```bash +sbatch omp_hello.sub +``` + +## View Results + +View the results from one of the sample OpenMP programs about task parallelism: + +```bash +cat omp_hello.sub.omyjobid +``` + +Example output: + +```text +SERIAL REGION: Runhost:negishi.a003.rcac.purdue.edu Thread:0 of 1 thread hello, world +PARALLEL REGION: Runhost:negishi.a003.rcac.purdue.edu Thread:0 of 128 threads hello, world +PARALLEL REGION: Runhost:negishi.a003.rcac.purdue.edu Thread:1 of 128 threads hello, world + ... +``` + +If the job failed to run, view error messages in the file `slurm-myjobid.out`. + +If an OpenMP program uses a lot of memory and 128 threads use all of the memory of the compute node, use fewer processor cores, or OpenMP threads, on that compute node. + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/python.md b/docs/userguides/negishi/run_jobs/python.md new file mode 100644 index 00000000..39857e4a --- /dev/null +++ b/docs/userguides/negishi/run_jobs/python.md @@ -0,0 +1,35 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Python examples on Negishi + +Python is a high-level, general-purpose, interpreted, dynamic programming language. We suggest using Anaconda which is a Python distribution made for large-scale data processing, predictive analytics, and scientific computing. For example, to use the default Anaconda distribution: + +```bash +$ module load conda +``` + +For a full list of available Anaconda and Python modules enter: + +```bash +$ module spider conda +``` + +- [**Example Python Jobs**](python/example_python_job.md) +- [**Managing Environments with Conda**](python/conda.md) +- [**Managing Packages with Pip**](python/pip.md) +- [**Installing Packages**](python/packages.md) +- [**Installing Packages from Source**](python/source.md) +- [**Example: Create and Use Biopython Environment with Conda**](python/environment_example.md) +- [**Numpy Parallel Behavior**](python/numpy.md) + + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/python/conda.md b/docs/userguides/negishi/run_jobs/python/conda.md new file mode 100644 index 00000000..500f9565 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/python/conda.md @@ -0,0 +1,87 @@ +--- +tags: + - Negishi + - Python +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Managing Environments with Conda + +Conda is a package manager in Anaconda that allows you to create and manage multiple environments where you can pick and choose which packages you want to use. To use Conda you must load an Anaconda module: + +``` +$ module load conda +``` + +Many packages are pre-installed in the global environment. To see these packages: + +``` +$ conda list +``` + +To create your own custom environment: + +``` +$ conda create --name MyEnvName python=3.8 FirstPackageName SecondPackageName -y +``` + +The `--name` option specifies that the environment created will be named MyEnvName. You can include as many packages as you require separated by a space. Including the `-y` option lets you skip the prompt to install the package. By default environments are created and stored in the $HOME/.conda directory. + +To create an environment at a custom location: + +``` +$ conda create --prefix=$HOME/MyEnvName python=3.8 PackageName -y +``` + +To see a list of your environments: + +``` +$ conda env list +``` + +To remove unwanted environments: + +``` +$ conda remove --name MyEnvName --all +``` + +To remove a package from an environment: + +``` +$ conda remove --name MyEnvName PackageName +``` + +Installing packages when creating your environment, instead of one at a time, will help you avoid dependency issues. + +To activate or deactivate an environment you have created: + +``` +$ source activate MyEnvName +$ source deactivate MyEnvName +``` + +If you created your conda environment at a custom location using `--prefix` option, then you can activate or deactivate it using the full path. + +``` +$ source activate $HOME/MyEnvName +$ source deactivate $HOME/MyEnvName +``` + +To use a custom environment inside a job you must load the module and activate the environment inside your job submission script. Add the following lines to your submission script: + +``` +$ module load conda +$ source activate MyEnvName +``` + +For more information about Python: + +* [The Python Programming Language - Official Website](http://www.python.org/) +* [Anaconda Python Distribution - Official Website](https://store.continuum.io/cshop/anaconda/) +* [Conda User Guide](https://conda.io/projects/conda/en/latest/user-guide/) + +[**Back to the Python section**](../python.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/python/environment_example.md b/docs/userguides/negishi/run_jobs/python/environment_example.md new file mode 100644 index 00000000..5d515d16 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/python/environment_example.md @@ -0,0 +1,65 @@ +--- +tags: + - Negishi + - Python +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + + +# Example: Create and Use Biopython Environment with Conda + +## Using conda to create an environment that uses the biopython package + +To use Conda you must first load the anaconda module: + +``` +module load conda +``` + +Create an empty conda environment to install biopython: + +``` +conda-env-mod create -n biopython +``` + +Now activate the biopython environment: + +``` +module load use.own +module load conda-env/biopython-py3.12.5 +``` + +Install the biopython packages in your environment: + +``` +conda install --channel anaconda biopython -y +Fetching package metadata .......... +Solving package specifications ......... +....... +Linking packages ... +[ COMPLETE ]|################################################################ +``` + +The `--channel` option specifies that it searches the anaconda channel for the biopython package. The `-y` argument is optional and allows you to skip the installation prompt. A list of packages will be displayed as they are installed. + +Remember to add the following lines to your job submission script to use the custom environment in your jobs: + +``` +module load conda +module load use.own +module load conda-env/biopython-py3.12.5 +``` + +If you need further help or run into any issues with creating environments, contact us. + +For more information about Python: + +* [The Python Programming Language - Official Website](http://www.python.org/) +* [Anaconda Python Distribution - Official Website](https://store.continuum.io/cshop/anaconda/) +* [Conda User Guide](https://conda.io/projects/conda/en/latest/user-guide/) + +[**Back to the Python section**](../python.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/python/example_python_job.md b/docs/userguides/negishi/run_jobs/python/example_python_job.md new file mode 100644 index 00000000..ac86d742 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/python/example_python_job.md @@ -0,0 +1,98 @@ +--- +tags: + - Negishi + - Python +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Example Python Jobs + +## Example 1: Hello world + +Prepare a Python input file with an appropriate filename, here named ```hello.py```: + +```bash +# FILENAME: hello.py + +import string, sys +print("Hello, world!") +``` + +Prepare a job submission file with an appropriate filename, here named ```myjob.sub```: + +```bash +#!/bin/bash +# FILENAME: myjob.sub + +module load conda + +python hello.py +``` + +Then, submit your job via SLURM and view the output file, which should simply output: + +```bash +Hello, world! +``` + +## Example 2: Matrix multiply + +Save the following script as ```matrix.py```: + +```bash +# Matrix multiplication program + +x = [[3,1,4],[1,5,9],[2,6,5]] +y = [[3,5,8,9],[7,9,3,2],[3,8,4,6]] + +result = [[sum(a*b for a,b in zip(x_row,y_col)) for y_col in zip(*y)] for x_row in x] + +for r in result: + print(r) +``` + +Change the last line in the job submission file above to read: + +```bash +python matrix.py +``` + +The standard output file from this job will result in the following matrix: + +```bash +[28, 56, 43, 53] +[65, 122, 59, 73] +[63, 104, 54, 60] +``` + +## Example 3: Sine wave plot using numpy and matplotlib packages + +Save the following script as ```sine.py```: + +```bash +import numpy as np +import matplotlib +matplotlib.use('Agg') +import matplotlib.pyplot as plt + +x = np.linspace(-np.pi, np.pi, 201) +plt.plot(x, np.sin(x)) +plt.xlabel('Angle [rad]') +plt.ylabel('sin(x)') +plt.axis('tight') +plt.savefig('sine.png') +``` + +Change your job submission file to submit this script and the job will output a png file and blank standard output and error files. + +For more information about Python: + +- [**The Python Programming Language - Official Website**](https://www.python.org/) +- [**Anaconda Python Distribution - Official Website**](https://store.continuum.io/cshop/anaconda/) +- [**Conda User Guide**](https://conda.io/projects/conda/en/latest/user-guide/) + +[**Back to the Python**](../python.md) diff --git a/docs/userguides/negishi/run_jobs/python/numpy.md b/docs/userguides/negishi/run_jobs/python/numpy.md new file mode 100644 index 00000000..c5d542c5 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/python/numpy.md @@ -0,0 +1,39 @@ +--- +tags: + - Negishi + - Python +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Numpy Parallel Behavior + +The widely available Numpy package is the best way to handle numerical computation in Python. The `numpy` package provided by our `anaconda` modules is optimized using Intel's MKL library. It will automatically parallelize many operations to make use of all the cores available on a machine. + +In many contexts that would be the ideal behavior. On the cluster however that very likely is not in fact the preferred behavior because often more than one user is present on the system and/or more than one job on a node. Having multiple processes contend for those resources will actually result in lesser performance. + +Setting the `MKL_NUM_THREADS` or `OMP_NUM_THREADS` environment variable(s) allows you to control this behavior. Our anaconda modules automatically set these variables to 1 *if and only if* you do not currently have that variable defined. + +When submitting batch jobs it is always a good idea to be explicit rather than implicit. If you are submitting a job that you want to make use of the full resources available on the node, set one or both of these variables to the number of cores you want to allow numpy to make use of. + +``` +#!/bin/bash + +module load conda +export MKL_NUM_THREADS=128 +... +``` + +If you are submitting multiple jobs that you intend to be scheduled together on the same node, it is probably best to restrict numpy to a single core. + +``` +#!/bin/bash + +module load conda +export MKL_NUM_THREADS=1 +``` + +[**Back to the Python section**](../python.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/python/packages.md b/docs/userguides/negishi/run_jobs/python/packages.md new file mode 100644 index 00000000..41004ade --- /dev/null +++ b/docs/userguides/negishi/run_jobs/python/packages.md @@ -0,0 +1,307 @@ +--- +tags: + - Negishi + - Python +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Installing Packages + +Installing Python packages in an Anaconda environment is recommended. One key advantage of Anaconda is that it allows users to install unrelated packages in separate self-contained environments. Individual packages can later be reinstalled or updated without impacting others. If you are unfamiliar with Conda environments, please check our [Conda Guide](../conda). + +To facilitate the process of creating and using Conda environments, we support a script (`conda-env-mod`) that generates a module file for an environment, as well as an optional Jupyter kernel to use this environment in a JupyterHub notebook. + +You must load one of the `anaconda` modules in order to use this script. + +``` +$ module load conda +``` + +Step-by-step instructions for installing custom Python packages are presented below. + +## Step 1: Create a conda environment + +Users can use the `conda-env-mod` script to create an empty conda environment. This script needs either a name or a path for the desired environment. After the environment is created, it generates a module file for using it in future. Please note that `conda-env-mod` is different from the official `conda-env` script and supports a limited set of subcommands. Detailed instructions for using `conda-env-mod` can be found with the command `conda-env-mod --help`. + +* **Example 1:** Create a conda environment named `mypackages` in user's `$HOME` directory. + + ``` + $ conda-env-mod create -n mypackages + ``` + +* **Example 2:** Create a conda environment named `mypackages` at a custom location. + + ``` + $ conda-env-mod create -p /depot/mylab/apps/mypackages + ``` + + Please follow the on-screen instructions while the environment is being created. After finishing, the script will print the instructions to use this environment. + + ``` + ... ... ... + Preparing transaction: ...working... done + Verifying transaction: ...working... done + Executing transaction: ...working... done + +------------------------------------------------------+ + | To use this environment, load the following modules: | + | module load use.own | + | module load conda-env/mypackages-py3.8.5 | + +------------------------------------------------------+ + Your environment "mypackages" was created successfully. + ``` + +Note down the module names, as you will need to load these modules every time you want to use this environment. You may also want to add the `module load` lines in your jobscript, if it depends on custom Python packages. + +By default, module files are generated in your `$HOME/privatemodules` directory. The location of module files can be customized by specifying the `-m /path/to/modules` option to `conda-env-mod`. + +**Note:** The main differences between `-p` and `-m` are: 1) `-p` will change the location of packages to be installed for the env and the module file will still be located at the `$HOME/privatemodules` directory as defined in use.own. 2) `-m` will only change the location of the module file. So the method to load modules created with `-m` and `-p` are different, see Example 3 for details. + +* **Example 3:** Create a conda environment named `labpackages` in your group's Data Depot space and place the module file at a shared location for the group to use. + + ``` + $ conda-env-mod create -p /depot/mylab/apps/labpackages -m /depot/mylab/etc/modules + ... ... ... + Preparing transaction: ...working... done + Verifying transaction: ...working... done + Executing transaction: ...working... done + +-------------------------------------------------------+ + | To use this environment, load the following modules: | + | module use /depot/mylab/etc/modules | + | module load conda-env/labpackages-py3.8.5 | + +-------------------------------------------------------+ + Your environment "labpackages" was created successfully. + ``` + +If you used a custom module file location, you need to run the `module use` command as printed by the command output above. + +By default, only the environment and a module file are created (no Jupyter kernel). If you plan to use your environment in a JupyterHub notebook, you need to append a `--jupyter` flag to the above commands. + +* **Example 4:** Create a Jupyter-enabled conda environment named `labpackages` in your group's Data Depot space and place the module file at a shared location for the group to use. + + ``` + $ conda-env-mod create -p /depot/mylab/apps/labpackages -m /depot/mylab/etc/modules --jupyter + ... ... ... + Jupyter kernel created: "Python (My labpackages Kernel)" + ... ... ... + Your environment "labpackages" was created successfully. + ``` + +## Step 2: Load the conda environment + + +* The following instructions assume that you have used `conda-env-mod` script to create an environment named `mypackages` (Examples 1 or 2 above). If you used `conda create` instead, please use `conda activate mypackages`. + + ``` + $ module load use.own + $ module load conda-env/mypackages-py3.8.5 + ``` + + Note that the `conda-env` module name includes the Python version that it supports (Python 3.8.5 in this example). This is same as the Python version in the `conda` module. +* If you used a custom module file location (Example 3 above), please use `module use` to load the `conda-env` module. + + ``` + $ module use /depot/mylab/etc/modules + $ module load conda-env/labpackages-py3.8.5 + ``` + +## Step 3: Install packages + + +Now you can install custom packages in the environment using either `conda install` or `pip install`. + +### Installing with conda + +* **Example 1:** Install OpenCV (open-source computer vision library) using conda. + + ``` + $ conda install opencv + ``` + +* **Example 2:** Install a specific version of OpenCV using conda. + + ``` + $ conda install opencv=4.5.5 + ``` + +* **Example 3:** Install OpenCV from a specific anaconda channel. + + ``` + $ conda install -c anaconda opencv + ``` + +### Installing with pip + +* **Example 4:** Install pandas using pip. + + ``` + $ pip install pandas + ``` + +* **Example 5:** Install a specific version of pandas using pip. + + ``` + $ pip install pandas==1.4.3 + ``` + + Follow the on-screen instructions while the packages are being installed. If installation is successful, please proceed to the next section to test the packages. + +**Note:** Do **NOT** run Pip with the `--user` argument, as that will install packages in a different location and might mess up your account environment. + +## Step 4: Test the installed packages + +To use the installed Python packages, you must load the module for your conda environment. If you have not loaded the `conda-env` module, please do so following the instructions at the end of Step 1. + +``` +$ module load use.own +$ module load conda-env/mypackages-py3.8.5 +``` + +* **Example 1:** Test that OpenCV is available. + + ``` + $ python -c "import cv2; print(cv2.__version__)" + ``` + +* **Example 2:** Test that pandas is available. + + ``` + $ python -c "import pandas; print(pandas.__version__)" + ``` + +If the commands finished without errors, then the installed packages can be used in your program. + +## Additional capabilities of conda-env-mod script + + +The `conda-env-mod` tool is intended to facilitate creation of a minimal Anaconda environment, matching module file and optionally a Jupyter kernel. Once created, the environment can then be accessed via familiar `module load` command, tuned and expanded as necessary. Additionally, the script provides several auxiliary functions to help manage environments, module files and Jupyter kernels. + +General usage for the tool adheres to the following pattern: + +``` +$ conda-env-mod help +$ conda-env-mod [optional arguments] +``` + +where required arguments are one of + +* `-n|--name ENV_NAME` (name of the environment) +* `-p|--prefix ENV_PATH` (location of the environment) + +and optional arguments further modify behavior for specific actions (e.g. `-m` to specify alternative location for generated module files). + +Given a required name or prefix for an environment, the `conda-env-mod` script supports the following subcommands: + +* `create` - to create a new environment, its corresponding module file and optional Jupyter kernel. +* `delete` - to delete existing environment along with its module file and Jupyter kernel. +* `module` - to generate just the module file for a given existing environment. +* `kernel` - to generate just the Jupyter kernel for a given existing environment (note that the environment has to be created with a `--jupyter` option). +* `help` - to display script usage help. + +Using these subcommands, you can iteratively fine-tune your environments, module files and Jupyter kernels, as well as delete and re-create them with ease. Below we cover several commonly occurring scenarios. + +**Note:** When you try to use `conda-env-mod delete`, remember to include the arguments as you create the environment (i.e. `-p package_location` and/or `-m module_location`). + +### Generating module file for an existing environment + +If you already have an existing configured Anaconda environment and want to generate a module file for it, follow appropriate examples from **Step 1** above, but use the `module` subcommand instead of the `create` one. E.g. + +``` +$ conda-env-mod module -n mypackages +``` + +and follow printed instructions on how to load this module. With an optional `--jupyter` flag, a Jupyter kernel will also be generated. + +**Note that** the module name `mypackages` should be exactly the same with the older conda environment name. **Note also that** if you intend to proceed with a Jupyter kernel generation (via the `--jupyter` flag or a `kernel` subcommand later), you will have to ensure that your environment has `ipython` and `ipykernel` packages installed into it. To avoid this and other related complications, we highly recommend making a fresh environment using a suitable `conda-env-mod create .... --jupyter` command instead. + +### Generating Jupyter kernel for an existing environment + +If you already have an existing configured Anaconda environment and want to generate a Jupyter kernel file for it, you can use the `kernel` subcommand. E.g. + +``` +$ conda-env-mod kernel -n mypackages +``` + +This will add a `"Python (My mypackages Kernel)"` item to the dropdown list of available kernels upon your next login to the JupyterHub. + +Note that generated Jupiter kernels are always personal (i.e. each user has to make their own, even for shared environments). Note also that you (or the creator of the shared environment) will have to ensure that your environment has `ipython` and `ipykernel` packages installed into it. + +### Managing and using shared Python environments + +Here is a suggested workflow for a common group-shared Anaconda environment with Jupyter capabilities: + +**The PI or lab software manager:** + +* Creates the environment and module file (once): + + ``` + $ module purge + $ module load conda + $ conda-env-mod create -p /depot/mylab/apps/labpackages -m /depot/mylab/etc/modules --jupyter + ``` + +* Installs required Python packages into the environment (as many times as needed): + + ``` + $ module use /depot/mylab/etc/modules + $ module load conda-env/labpackages-py3.8.5 + $ conda install ....... # all the necessary packages + ``` + +**Lab members:** + +* Lab members can start using the environment in their command line scripts or batch jobs simply by loading the corresponding module: + + ``` + $ module use /depot/mylab/etc/modules + $ module load conda-env/labpackages-py3.8.5 + $ python my_data_processing_script.py ..... + ``` + +* To use the environment in Jupyter notebooks, each lab member will need to create his/her own Jupyter kernel (once). This is because Jupyter kernels are private to individuals, even for shared environments. + + ``` + $ module use /depot/mylab/etc/modules + $ module load conda-env/labpackages-py3.8.5 + $ conda-env-mod kernel -p /depot/mylab/apps/labpackages + ``` + +A similar process can be devised for instructor-provided or individually-managed class software, etc. + +## Troubleshooting + +* Python packages often fail to install or run due to dependency incompatibility with other packages. More specifically, if you previously installed packages in your home directory it is safer to clean those installations. + + ``` + $ mv ~/.local ~/.local.bak + $ mv ~/.cache ~/.cache.bak + ``` + +* Unload all the modules. + + ``` + $ module purge + ``` + +* Clean up PYTHONPATH. + + ``` + $ unset PYTHONPATH + ``` + +* Next load the modules (e.g. anaconda) that you need. + + ``` + $ module load conda/2024.02-py311 + $ module load use.own + $ module load conda-env/2024.02-py311 + ``` + +* Now try running your code again. + +* Few applications only run on specific versions of Python (e.g. Python 3.6). Please check the documentation of your application if that is the case. + +[**Back to the Python section**](../python.md) diff --git a/docs/userguides/negishi/run_jobs/python/pip.md b/docs/userguides/negishi/run_jobs/python/pip.md new file mode 100644 index 00000000..97277d3b --- /dev/null +++ b/docs/userguides/negishi/run_jobs/python/pip.md @@ -0,0 +1,59 @@ +--- +tags: + - Negishi + - Python +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- +# Managing Packages with Pip + +Pip is a Python package manager. Many Python package documentation provide `pip` instructions that result in permission errors because by default `pip` will install in a system-wide location and fail. + +``` + +Exception: +Traceback (most recent call last): +... ... stack trace ... ... +OSError: [Errno 13] Permission denied: '/apps/cent7/anaconda/2020.07-py38/lib/python3.8/site-packages/mkl_random-1.1.1.dist-info' +``` + +If you encounter this error, it means that you cannot modify the global Python installation. We recommend installing Python packages in a conda environment. Detailed instructions for installing packages with `pip` can be found in our [Python package installation page](../packages). + +Below we list some other useful `pip` commands. + +* Search for a package in PyPI channels: + + ``` + $ pip search packageName + ``` +* Check which packages are installed globally: + + ``` + $ pip list + ``` +* Check which packages you have personally installed: + + ``` + $ pip list --user + ``` +* Snapshot installed packages: + + ``` + $ pip freeze > requirements.txt + ``` +* You can install packages from a snapshot inside a new conda environment. Make sure to load the appropriate conda environment first. + + ``` + $ pip install -r requirements.txt + ``` + +For more information about Python: + +* [The Python Programming Language - Official Website](http://www.python.org/) +* [Anaconda Python Distribution - Official Website](https://store.continuum.io/cshop/anaconda/) +* [Conda User Guide](https://conda.io/projects/conda/en/latest/user-guide/) + +[**Back to the Python section**](../python.md) diff --git a/docs/userguides/negishi/run_jobs/python/source.md b/docs/userguides/negishi/run_jobs/python/source.md new file mode 100644 index 00000000..76120788 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/python/source.md @@ -0,0 +1,64 @@ +--- +tags: + - Negishi + - Python +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + + +# Installing Packages from Source + +We maintain several [Anaconda](https://www.anaconda.com/) installations. Anaconda maintains numerous popular scientific Python libraries in a single installation. If you need a Python library not included with normal Python we recommend first checking Anaconda. For a list of modules currently installed in the Anaconda Python distribution: + +``` +$ module load conda +$ conda list +# packages in environment at /apps/spack/negishi/apps/anaconda/2020.02-py37-gcc-4.8.5-u747gsx: +# +# Name Version Build Channel +_ipyw_jlab_nb_ext_conf 0.1.0 py37_0 +_libgcc_mutex 0.1 main +alabaster 0.7.12 py37_0 +anaconda 2020.02 py37_0 +... +``` + +If you see the library in the list, you can simply import it into your Python code after loading the Anaconda module. + +If you do not find the package you need, you should be able to install the library in your own Anaconda customization. First try to install it with [Conda or Pip](./packages.md). If the package is not available from either Conda or Pip, you may be able to install it from source. + +Use the following instructions as a guideline for installing packages from source. Make sure you have a download link to the software (usually it will be a `tar.gz` archive file). You will substitute it on the wget line below. + +We also assume that you have already created an empty conda environment as described in our [Python package installation guide](./packages.md). + +``` +$ mkdir ~/src +$ cd ~/src +$ wget http://path/to/source/tarball/app-1.0.tar.gz +$ tar xzvf app-1.0.tar.gz +$ cd app-1.0 +$ module load conda +$ module load use.own +$ module load conda-env/mypackages-py3.8.5 +$ python setup.py install +$ cd ~ +$ python +>>> import app +>>> quit() +``` + +The "import app" line should return without any output if installed successfully. You can then import the package in your python scripts. + +If you need further help or run into any issues installing a library, contact us. + +For more information about Python: + +* [The Python Programming Language - Official Website](http://www.python.org/) +* [Anaconda Python Distribution - Official Website](https://store.continuum.io/cshop/anaconda/) +* [Conda User Guide](https://conda.io/projects/conda/en/latest/user-guide/) + +[**Back to the Python section**](../python.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/queues.md b/docs/userguides/negishi/run_jobs/queues.md new file mode 100644 index 00000000..0118f1c4 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/queues.md @@ -0,0 +1,121 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Slurm accounts, partitions, and QOS options +## Queues + +On Negishi, the required options for job submission deviates from some of the other community clusters you might have experience using. In general every job submission will have four parts: “`sbatch --ntasks=1 --cpus-per-task=4 --partition=cpu --account=rcac --qos=standby`” + +1. The number and type of resources you want (`--ntasks=1 --cpus-per-task=4`) +2. The partition where the resources are located (`--partition=cpu`) +3. The account the resources should come out of ( `--account=rcac`) +4. The quality of service (QOS) this job expects from the resources (`--qos=standby`) + +Table Summary of Changes + +| Use Case | Old Syntax | New Syntax | +| --- | --- | --- | +| Submit a job to your group's account | `sbatch -A mygroup` | `sbatch -A mygroup -p cpu` | +| Submit a standby job | `sbatch -A standby` | `sbatch -A mygroup -p cpu -q standby` | +| Submit a highmem job | `sbatch -A highmem` | `sbatch -A mygroup -p highmem` | +| Submit a gpu job | `sbatch -A gpu` | `sbatch -A mygroup -p gpu` | +| Submit a job to your group's interactive account | `sbatch -A interactive` | `sbatch -A mygroup -p negishi -q interactive` | + +If you have used other clusters, you will be familiar with the first item. If you have not, you can read about how to format the request [on our job submission page.](../submit_script) The rest of this page will focus on the last three items. + +## Partitions + +On Negishi, the various types of nodes on the cluster are organized into distinct partitions. This allows jobs to different node types to be charged separately and differently. This also means that Instead of only needing to specify the account name in the job script, the desired partition must also be specified. Each of these partitions is subject to different limitations and has a specific use case that will be described below.  + +### CPU Partition + +This partition contains the resources a group purchases access to when they purchase CPU resources on Negishi and is made up of 446 Bell-A nodes. Each of these nodes contains two Zen 3 AMD EPYC 7763 64-core processors for a total of 128 cores and 256 GB of memory for a total of more than 57,000 cores in the partition. Memory in this partition is allocated proportional to your core request such that each core is given about 2 GB of memory per core requested. Submission to this partition can be accomplished by using the option: `-p cpu` or `--partition=cpu`. + +The purchasing model for this partition allows groups to purchase high priority access to some number of cores. When an account uses resources in this account by submitting a job tagged with the `normal` QOS, the cores used by that job are withdrawn from the account and deposited back into the account when the job terminates. + +When using the CPU partition, jobs are tagged by the `normal` QOS by default, but they can be tagged with the `standby` QOS if explicitly submitted using the `-q standby` or `--qos=standby` option. + +1. Jobs tagged with the `normal` QOS are subject to the following policies: + 1. Jobs have a high priority and should not need to wait very long before starting. + 2. Any cores requested by these jobs are withdrawn from the account until the job terminates. + 3. These jobs can run for up to two weeks at a time. +2. Jobs tagged with the `standby` QOS are subject to the following policies: + 1. Jobs have a low priority and there is no expectation of job start time. If the partition is very busy with jobs using the `normal` QOS or if you are requesting a very large job, then jobs using the `standby` QOS may take hours or days to start. + 2. These jobs can use idle resources on the cluster and as such cores requested by these jobs are not withdrawn from the account to which they were submitted. + 3. These jobs can run for up to four hours at a time. + +Available QOSes: `normal`, `standby` + +### Highmem Partition + +This partition is made up of 6 Bell-B nodes which have four times as much memory as a standard Bell-A node, and access to this partition is given to all accounts on the cluster to enable work that has higher memory requirements. Each of these nodes contains two Zen 2 AMD EPYC 7763 64-core processors for a total of 128 cores and 1 TB of memory. Memory in this partition is allocated proportional to your core request such that each core is given about 8 GB of memory per core requested. Submission to this partition can be accomplished by using the option: `-p highmem` or `--partition=highmem`. + +When using the Highmem partition, jobs are tagged by the `normal` QOS by default, and this is the only QOS that is available for this partition, so there is no need to specify a QOS when using this partition. Additionally jobs are tagged by a highmem partition QOS that enforces the following policies + +1. There is no expectation of job start time as these nodes are a shared resources that are given as a bonus for purchasing access to high priority access to resources on Negishi +2. You can have 2 jobs running in this partition at once +3. You can have 8 jobs submitted to thie partition at once +4. Your jobs must use more than 64 of the 128 cores on the node otherwise your memory footprint would fit on a standard Negishi-A node +5. These jobs can run for up to 24 hours at a time. + +Available QOSes: `normal` + +### GPU Partition + +This partition is made up of 5 Negishi-G nodes. Each of these nodes contains two AMD MI210s and two Zen 2 AMD EPYC 7313 16-core processors for a total of 32 cores and 512GB of memory. Memory in this partition is allocated proportional to your core request such that each core is given about 8 GB of memory per core requested. You should request cores proportional to the number of GPUs you are using in this partition (i.e. if you only need one of the two GPUs, you should request half of the cores on the node) Submission to this partition can be accomplished by using the option: `-p gpu` or `--partition=gpu`. + +When using the gpu partition, jobs are tagged by the `normal` QOS by default, and this is the only QOS that is available for this partition, so there is no need to specify a QOS when using this partition. Additionally jobs are tagged by a gpu partition QOS that enforces the following policies + +1. There is no expectation of job start time as these nodes are a shared resources that are given as a bonus for purchasing access to high priority access to resources on Negishi +2. You can use up to 2 GPUs in this partition at once +3. You can have 8 jobs submitted to thie partition at once +4. These jobs can run for up to 24 hours at a time. + +Available QOSes: `normal` + +### Login Partition + +This partition contains the resources a group purchases access to when they purchase "interactive access" on Negishi. Interactive access allows submission of jobs directly to the front ends for immediate job start times. These jobs can only request up to 4 CPUs and 8 GB of memory each and interactive users can only have one job at a time. Submission to this partition can be accomplished by using the option: `-p login` or `--partition=login`. In order to use this partition, you must submit using the interactive QOS which enforces the following policies: + +1. You can have one job running at a time. +2. You can use up to 4 cores and 8 GB of memory at a time. +3. Jobs can run for up to 24 hours. + +Available QOSes: `interactive` + +## Accounts + + +On the Negishi community cluster, users will have access to one or more accounts, also known as queues. These accounts are dedicated to and named after each partner who has purchased access to the cluster, and they provide partners and their researchers with priority access to their portion of the cluster. These accounts can be thought of as bank accounts that contain the resources a group has purchased access to which may include some number of cores. To see the list of accounts that you have access to on Negishi as well as the resources they contain, you can use the command `slist`. + +On Negishi, you must explicitly define the account that you want to submit to using the `-A`or`--account=` option. + +## Quality of Service (QOS) + +On Negishi, we use a Slurm concept called a Quality of Service or a QOS. A QOS can be thought of as a tag for a job that tells the scheduler how that job should be treated with respect to limits, priority, etc. The cluster administrators define the available QOSes as well as the policies for how each QOS should be treated on the cluster. A toy example of such a policy may be "no single user can have more than 200 jobs that has been tagged with a QOS named *highpriority*". + +There are two classes of QOSes and a job can have both: + +1. Partition QOSes: A partition QOS is a tag that is automatically added to your job when you submit to a partition that defines a partition QOS. +2. Job QOSes: A Job QOS is a tag that you explicitly give to a job using the option `-q`or`--qos=`. By explicitly tagging your jobs this way, you can choose the policy that each one of your jobs should abide by. We will describe the policies for the available job QOSes in the partition section below. + +As an extended metaphor, if we think of a job as a package that we need to have shipped to some destination, then the partition can be thought of as the carrier we decide to ship our package with. That carrier is going to have some company policies that dictate how you need to label/pack that package, and that company policy is like the partition QOS. It is the policy that is enforced for simply deciding to use that carrier, or in this case, deciding to submit to a particular partition. + +The Job QOS can then be thought of as the various different types of shipping options that carrier might offer. You might pay extra to have that package shipped overnight. On the other hand you may choose to pay less and have your package arrive as available. Once we decide to go with a particular carrier, we are subject to their company policy, but we also have some degree of control through choosing one of their available shipping options. In the same way, when you choose to submit to a partition, you are subject to the limits enforced by the partition QOS, but you may be able to ask for your job to be handled a particular way by specifying a job QOS offered by the partition. + +In order for a job to use a Job QOS, the user submitting the job must have access to the QOS, the account the job is being submitted to must accept the QOS, and the partition the job is being submitted to must accept the QOS. The below list of job QOSes are QOSes that every user and every account of Negishi has access to: + +1. `normal`: The `normal` QOS is the default job QOS on the cluster meaning if you do not explicitly list an alternative job QOS, your job will be tagged with this QOS. The policy for this QOS provides a high priority and does not add any additional limits. +2. `standby`: The `standby` QOS must be explicitly used if desired by using the option `-q standby` or `--qos=standby`. The policy for this QOS gives access to idle resources on the cluster. Jobs tagged with this QOS are "low priority" jobs and are only allowed to run for up to four hours at a time, however the resources used by these jobs do not count against the resources in your Account. For users of our previous clusters, usage of this QOS replaces the previous `-A standby` style of submission. + +Some of these QOSes may not be available in every partition. Each of the partitions in the following section will enumerate which of these QOSes are allowed in the partition. + + +[**Back to the Running Jobs section**](index.md) \ No newline at end of file diff --git a/docs/userguides/negishi/run_jobs/r.md b/docs/userguides/negishi/run_jobs/r.md new file mode 100644 index 00000000..3fb0eef4 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/r.md @@ -0,0 +1,23 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# R examples on Negishi + +R, a GNU project, is a language and environment for data manipulation, statistics, and graphics. It is an open source version of the S programming language. R is quickly becoming the language of choice for data science due to the ease with which it can produce high quality plots and data visualizations. It is a versatile platform with a large, growing community and collection of packages. + +For more general information on R visit [The R Project for Statistical Computing](https://www.r-project.org/) + +- [**Loading Data into R**](r/example_loading_into_r.md) +- [**Installing R Packages**](r/example_installing_r_packages.md) +- [**RStudio**](r/example_rstudio.md) +- [**Running R Jobs**](r/example_running_r_jobs.md) +- [**Setting Up R Preferences with .Rprofile**](r/example_r_profile_setup.md) + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/r/example_installing_r_packages.md b/docs/userguides/negishi/run_jobs/r/example_installing_r_packages.md new file mode 100644 index 00000000..308fd0a9 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/r/example_installing_r_packages.md @@ -0,0 +1,116 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Installing R Packages + +## Challenges of Managing R Packages in the Cluster Environment + +- Different clusters have different hardware and softwares. So, if you have access to multiple clusters, you must install your R packages separately for each cluster. +- Each cluster has multiple versions of R and packages installed with one version of R may not work with another version of R. So, libraries for each R version must be installed in a separate directory. +- You can define the directory where your R packages will be installed using the environment variable `R_LIBS_USER`. +- For your convenience, a sample [~/.Rprofile example file](/assets/scripts/userguides/Rprofile_example) is provided that can be downloaded to your cluster account and renamed into `~/.Rprofile` (or appended to one) to customize your installation preferences. [Detailed instructions here](example_r_profile_setup.md). + +## Installing Packages +- ### Step 0: Setup Installation Preferences +Follow the [steps for setting up your](example_r_profile_setup.md) `~/.Rprofile` preferences. This step needs to be done only once. If you have created a `~/.Rprofile` file previously on Negishi, ignore this step. + +- ### Step 1: Check if the package is already installed +As part of the R installations on community clusters, a lot of R libraries are pre-installed. You can check if your package is already installed by opening an R terminal and entering the command ```installed.packages()```. For example, + +```r +module load r/4.4.1 +R +``` + +``` +installed.packages()["units",c("Package","Version")] +Package Version +"units" "0.8-1" +quit() +``` + +If the package you are trying to use is already installed, simply load the library, e.g., `library('units')`. Otherwise, move to the next step to install the package. + +- ### Step 2: Load required dependencies (if needed) +For simple packages you may not need this step. However, some R packages depend on other libraries. For example, the ```sf``` package depends on ```gdal``` and ```geos``` libraries. So, you will need to load the corresponding modules before installing ```sf```. Read the documentation for the package to identify which modules should be loaded. + +```r +module load gdal +module load geos +``` + +- ### Step 3: Install the package +Now install the desired package using the command ```install.packages('package_name')```. R will automatically download the package and all its dependencies from [CRAN](https://cran.r-project.org/mirrors.html) and install each one. Your terminal will show the build progress and eventually show whether the package was installed successfully or not. + +```r +R +``` +``` +install.packages('sf', repos="https://cran.case.edu/") +Installing package into ‘/home/myusername/R/x86_64-pc-linux-gnu-library/4.4.1’ +(as ‘lib’ is unspecified) +trying URL 'https://cran.case.edu/src/contrib/sf_0.9-7.tar.gz' +Content type 'application/x-gzip' length 4203095 bytes (4.0 MB) +================================================== +downloaded 4.0 MB +... +... +more progress messages +... +... +** testing if installed package can be loaded from final location +** testing if installed package keeps a record of temporary installation path +* DONE (sf) + +The downloaded source packages are in + ‘/tmp/RtmpSVAGio/downloaded_packages’ +``` + +- ### Step 4: Troubleshooting (if needed) +If Step 3 ended with an error, you need to investigate why the build failed. Most common reason for build failure is not loading the necessary modules. + +## Loading Libraries +Once you have packages installed you can load them with the ```library()``` function as shown below: + +```r +library('packagename') +``` +The package is now installed and loaded and ready to be used in R. + +## Example: Installing ```dplyr``` + +```r +module load r +R +``` + +``` +install.packages('dplyr', repos="http://ftp.ussg.iu.edu/CRAN/") +Installing package into ‘/home/myusername/R/negishi/4.4.1’ +(as ‘lib’ is unspecified) + ... +also installing the dependencies 'crayon', 'utf8', 'bindr', 'cli', 'pillar', 'assertthat', 'bindrcpp', 'glue', 'pkgconfig', 'rlang', 'Rcpp', 'tibble', 'BH', 'plogr' + ... + ... + ... +The downloaded source packages are in + '/tmp/RtmpHMzm9z/downloaded_packages' + +library(dplyr) + +Attaching package: 'dplyr' +``` + +For more information about installing R packages: + +- [Installing additional R packages on Linux](http://cran.r-project.org/doc/manuals/r-release/R-admin.html#Installing-packages) +- [List of Packages](https://cran.r-project.org/web/packages/) + +[**Back to the R Examples section**](../r.md) diff --git a/docs/userguides/negishi/run_jobs/r/example_loading_into_r.md b/docs/userguides/negishi/run_jobs/r/example_loading_into_r.md new file mode 100644 index 00000000..588c715c --- /dev/null +++ b/docs/userguides/negishi/run_jobs/r/example_loading_into_r.md @@ -0,0 +1,38 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Loading Data into R + +R is an environment for manipulating data. In order to manipulate data, it must be brought into the R environment. R has a function to read any file that data is stored in. Some of the most common file types like comma-separated variable(CSV) files have functions that come in the basic R packages. Other less common file types require additional packages to be installed. To read data from a CSV file into the R environment, enter the following command in the R prompt: + +```bash +> read.csv(file = "path/to/data.csv", header = TRUE) +``` + +When R reads the file it creates an object that can then become the target of other functions. By default the read.csv() function will give the object the name of the .csv file. To assign a different name to the object created by read.csv enter the following in the R prompt: + +```bash +> my_variable <- read.csv(file = "path/to/data.csv", header = FALSE) +``` + +To display the properties (structure) of loaded data, enter the following: + +```bash +> str(my_variable) +``` + +For more functions and tutorials: + +- [The R Manuals](https://cran.r-project.org/manuals.html) +- [Other R Examples](https://www.mayin.org/ajayshah/KB/R/index.html) +- [Software Carpentry - Programming with R](https://swcarpentry.github.io/r-novice-inflammation/) +- [Data Carpentry Lessons](http://www.datacarpentry.org/lessons/) + +[**Back to the R Examples section**](../r.md) diff --git a/docs/userguides/negishi/run_jobs/r/example_r_profile_setup.md b/docs/userguides/negishi/run_jobs/r/example_r_profile_setup.md new file mode 100644 index 00000000..1c897690 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/r/example_r_profile_setup.md @@ -0,0 +1,35 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +--- + +# Setting up R Preferences with .Rprofile + +For your convenience, a sample [~/.Rprofile example file](/assets/scripts/userguides/Rprofile_example) is provided that can be downloaded to your cluster account and renamed into ```~/.Rprofile``` (or appended to one). Follow these steps to download our recommended ```~/.Rprofile``` example and copy it into place: + +```bash +curl -#LO https://docs.rcac.purdue.edu/assets/scripts/userguides/Rprofile_example +mv -ib Rprofile_example ~/.Rprofile +``` + +The above installation step needs to be done only once on Negishi. Now load the R module and run R: + +```bash +module load r/4.4.1 +R +``` + +```R +.libPaths() +[1] "/home/{user}/R/negishi/4.1.2-gcc-6.3.0-ymdumss" +[2] "/apps/spack/negishi/apps/r/4.1.2-gcc-6.3.0-ymdumss/rlib/R/library" +``` + +```.libPaths()``` should output something similar to above if it is set up correctly. + +You are now ready to install R packages into the dedicated directory `/home/myusername/R/negishi/4.1.2-gcc-6.3.0-ymdumss`. + +[**Back to the Installing R Packages section**](example_installing_r_packages.md) diff --git a/docs/userguides/negishi/run_jobs/r/example_rstudio.md b/docs/userguides/negishi/run_jobs/r/example_rstudio.md new file mode 100644 index 00000000..b7b624fa --- /dev/null +++ b/docs/userguides/negishi/run_jobs/r/example_rstudio.md @@ -0,0 +1,42 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# RStudio + +RStudio is a graphical integrated development environment (IDE) for R. RStudio is the most popular environment for developing both R scripts and packages. RStudio is provided on most Research systems. + +There are two methods to launch RStudio on the cluster: command-line and application menu icon. + +## Launch RStudio via command-line: + +```bash +module load gcc +module load r +module load rstudio +rstudio +``` + +Note that RStudio is a graphical program and in order to run it you must have a local X11 server running or use [Thinlinc](../../accounts.md#thinlinc) Remote Desktop environment. See the [ssh X11 forwarding section](../../accounts.md#ssh-x11-forwarding) for more details. + +## Launch Rstudio via the application menu icon: + +- Log into desktop.negishi.rcac.purdue.edu with web browser or [ThinLinc](../../accounts.md#thinlinc) client +- Click on the ```Applications``` drop down menu on the top left corner +- Choose ```Cluster Software``` and then ```RStudio``` + +![This image displays the Thinlinc Application Launcher menu and depicts the user selecting Cluster Software > Rstudio](example_rstudio_app_menu.png) + +R and RStudio are free to download and run on your local machine. For more information about RStudio: + +- [RStudio Official Website](https://www.rstudio.com/) +- [RStudio Essentials: Tutorial](https://www.rstudio.com/resources/webinars/#rstudioessentials) +- [DataCamp: Working with the RStudio IDE](https://www.datacamp.com/courses/working-with-the-rstudio-ide-part-1) + +[**Back to the R Examples section**](../r.md) diff --git a/docs/userguides/negishi/run_jobs/r/example_rstudio_app_menu.png b/docs/userguides/negishi/run_jobs/r/example_rstudio_app_menu.png new file mode 100644 index 00000000..17992cba Binary files /dev/null and b/docs/userguides/negishi/run_jobs/r/example_rstudio_app_menu.png differ diff --git a/docs/userguides/negishi/run_jobs/r/example_running_r_jobs.md b/docs/userguides/negishi/run_jobs/r/example_running_r_jobs.md new file mode 100644 index 00000000..3ecd898f --- /dev/null +++ b/docs/userguides/negishi/run_jobs/r/example_running_r_jobs.md @@ -0,0 +1,49 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Running R Jobs + +This section illustrates how to submit a small R job to a SLURM queue. The example job computes a Pythagorean triple. + +Prepare an R input file with an appropriate filename, here named ```myjob.R```: + +```bash +# FILENAME: myjob.R + +# Compute a Pythagorean triple. +a = 3 +b = 4 +c = sqrt(a*a + b*b) +c # display result +``` + +Prepare a job submission file with an appropriate filename, here named ```myjob.sub```: + +```bash +#!/bin/bash +# FILENAME: myjob.sub + +module load r + +# --vanilla: +# --no-save: do not save datasets at the end of an R session +R --vanilla --no-save < myjob.R +``` + +[Submit the Job and view the results](/userguides/negishi/run_jobs/) + +For other examples or R jobs: + +- [The R Manuals](http://cran.r-project.org/manuals.html) +- [Other R Examples](http://www.mayin.org/ajayshah/KB/R/index.html) +- [Software Carpentry - Programming with R](https://swcarpentry.github.io/r-novice-inflammation/) +- [Data Carpentry Lessons](http://www.datacarpentry.org/lessons/) + +[**Back to the R Examples section**](../r.md) diff --git a/docs/userguides/negishi/run_jobs/serial_jobs.md b/docs/userguides/negishi/run_jobs/serial_jobs.md new file mode 100644 index 00000000..6214eaf1 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/serial_jobs.md @@ -0,0 +1,41 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Running Serial Jobs + +This shows how to submit one of the serial programs compiled in the section [Compiling Serial Programs](../../compile/serial). + +Create a job submission file: + +```bash +#!/bin/bash +# FILENAME: serial_hello.sub + +./serial_hello +``` + +Submit the job: + +```bash +sbatch --nodes=1 --ntasks=1 --time=00:01:00 serial_hello.sub +``` + +After the job completes, view results in the output file: + +```bash +cat slurm-myjobid.out + +Runhost:a009.negishi.rcac.purdue.edu +hello, world +``` + +If the job failed to run, then view error messages in the file ```slurm-myjobid.out```. + +[**Back to the Example Jobs section**](generic_slurm_jobs.md) diff --git a/docs/userguides/negishi/run_jobs/simple_job.md b/docs/userguides/negishi/run_jobs/simple_job.md new file mode 100644 index 00000000..8fc2eb9e --- /dev/null +++ b/docs/userguides/negishi/run_jobs/simple_job.md @@ -0,0 +1,52 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Simple Jobs + +Every SLURM job consists of a job submission file. A job submission file contains a list of commands that run your program and a set of resource (nodes, walltime, queue) requests. The resource requests can appear in the job submission file or can be specified at submit-time as shown below. + +This simple example submits the job submission file `hello.sub` to the `standby` queue on Negishi and requests a single node: + +```bash +#!/bin/bash +# FILENAME: hello.sub + +# Show this ran on a compute node by running the hostname command. +hostname + +echo "Hello World" +``` +``` +sbatch -A standby --nodes=1 --ntasks=1 --cpus-per-task=1 --gpus-per-node=1 --time=00:01:00 hello.sub +Submitted batch job 3521 +``` + +For a real job you would replace ```echo "Hello World"``` with a command, or sequence of commands, that run your program. + +After your job finishes running, the ```ls``` command will show a new file in your directory, the ```.out``` file: + +```bash +ls -l +hello.sub +slurm-3521.out +``` + +The file ```slurm-3521.out``` contains the output and errors your program would have written to the screen if you had typed its commands at a command prompt: + +```bash +cat slurm-3521.out + +a001.negishi.rcac.purdue.edu +Hello World +``` + +You should see the hostname of the compute node your job was executed on. Following should be the "Hello World" statement. + +[**Back to the Example Jobs section**](generic_slurm_jobs.md) diff --git a/docs/userguides/negishi/run_jobs/specific_nodes.md b/docs/userguides/negishi/run_jobs/specific_nodes.md new file mode 100644 index 00000000..277c8d04 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/specific_nodes.md @@ -0,0 +1,36 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + + +# Specific Types of Nodes + +SLURM allows running a job on [specific types of compute nodes](../overview.md) to accommodate special hardware requirements (e.g. a certain CPU or GPU type, etc.) + +Cluster nodes have a set of descriptive features assigned to them, and users can specify which of these features are required by their job by using the constraint option at submission time. Only nodes having features matching the job constraints will be used to satisfy the request. + +**Example:** a job requires a compute node in an "A" sub-cluster: + + +``` +sbatch --nodes=1 --ntasks=128 --constraint=A myjobsubmissionfile.sub +``` + +Compute node allocated: + +``` +a003.negishi +``` + +Feature constraints can be used for both batch and interactive jobs, as well as for individual job steps inside a job. Multiple constraints can be specified with a predefined syntax to achieve complex request logic (see detailed description of the '--constraint' option in `man sbatch` or online Slurm documentation). + +Refer to [Detailed Hardware Specification](../overview.md) section for list of available sub-cluster labels, their respective per-node memory sizes and other hardware details. You could also use `sfeatures` command to list available constraint feature names for different node types. + + +[**Back to the Example Jobs section**](generic_slurm_jobs.md) diff --git a/docs/userguides/negishi/run_jobs/submit_script.md b/docs/userguides/negishi/run_jobs/submit_script.md new file mode 100644 index 00000000..88329cfd --- /dev/null +++ b/docs/userguides/negishi/run_jobs/submit_script.md @@ -0,0 +1,118 @@ +--- +tags: + - Negishi +authors: + - hkashgar +cluster: Negishi +search: + boost: 2 +--- + +# Submitting a Job + +Once you have a [job submission file](./creating_the_submission_script.md), you may submit this script to SLURM using the `sbatch` command. SLURM will find, or wait for, available resources matching your request and run your job there. + +​On Negishi, in order to submit jobs, you need to specify the partition, account and Quality of Service (QoS) name to which you want to submit your jobs. To familiarize yourself with the partitions and QoS available on Negishi, visit [Negishi Queues and Partitions](./queues.md). To check the available partitions on Negishi, you can use the `showpartitions` , and to check your available accounts you can use `slist` commands. Slurm uses the term "Account" with the option `-A` or `--account=` to specify different batch accounts, the option `-p` or `--partition=` to select a specific partition for job submission, and the option `-q` or `--qos=` . + +``` +Partition statistics for cluster negishi at Wed Jul 30 13:07:17 EDT 2025 + + Partition #Nodes #CPU_cores Cores_pending Job_Nodes MaxJobTime Cores Mem/Node + Name State Total Idle Total Idle Resorc Other Min Max Day-hr:mn /node (GB) + negishi-nodes up$ 6 0 768 768 0 0 1 infin infinite 128 257 + cpu up$ 446 0 57088 57067 0 163 1 infin infinite 128 257 + highmem up$ 6 0 768 768 0 0 1 infin infinite 128 1031 + cms up$ 16 0 4096 3840 0 9036 1 infin infinite 256 515 + gpu up 5 0 160 0 0 30 1 infin infinite 32 515 + negishi-login up$ 8 0 2048 2048 0 0 1 infin infinite 256 515 +negishi-profiling up 4 3 512 512 0 0 1 infin infinite 128 257 +``` + +### CPU Partition + +The CPU partition on Negishi has two Quality of Service (QoS) levels: **normal** and **standby**. To submit your job to one compute node on `cpu` partition and 'normal' QoS which has "high priority": + +``` +$ sbatch --nodes=1 --ntasks=1 --partition=cpu --account=accountname --qos=normal myjobsubmissionfile +$ sbatch -N1 -n1 -p cpu -A accountname -q normal myjobsubmissionfile +``` + + To submit your job to one compute node on `cpu` partition and 'standby' QoS which is has "low priority": + +``` +$ sbatch --nodes=1 --ntasks=1 --partition=cpu --account=accountname --qos=standby myjobsubmissionfile +$ sbatch -N1 -n1 -p cpu -A accountname -q standby myjobsubmissionfile +``` + +### GPU Partition + +On the GPU partition on **Negishi** you don’t need to specify the QoS name because only one QoS exists for this partition, and the default is **normal**. To submit your job to one compute node requesting one GPU on the `gpu` partition under the 'normal' QoS which has "high priority": + +``` +$ sbatch --nodes=1 --gpus-per-node=1 --ntasks=1 --cpus-per-task=64 --partition=gpu --account=accountname myjobsubmissionfile +$ sbatch -N1 --gpus-per-node=1 -n1 -c64 -p gpu -A accountname -q normal myjobsubmissionfile +``` + +### Highmem Partition + +To submit your job to a compute node in the highmem partition, you don’t need to specify the QoS name because only one QoS exists for this partition, and the default is **normal**. However, the **highmem** partition is *only* suitable for jobs with memory requirements that exceed the capacity of a standard node, so the number of requested tasks should be appropriately high. + +``` +$ sbatch --nodes=1 --ntasks=1 --cpus-per-task=64 --partition=highmem --account=accountname myjobsubmissionfile +$ sbatch -N1 -n1 -c64 -p gpu -A accountname myjobsubmissionfile +``` + +### Login Partition + +The login partition is reserved for Negishi users who have purchased "interactive" access to Negishi. As an access control, in order to submit to this partition, you must supply the **interactive**QoS as a job option as this is the only QoS accepted by the partition. This partition is meant to give near-immediate start times to computationally modest jobs. + +``` +$ sbatch --nodes=1 --ntasks=1 --cpus-per-task=4 --partition=login --qos=interactive --account=accountname myjobsubmissionfile +$ sbatch -N1 -n1 -c4 -p gpu -q interactive -A accountname myjobsubmissionfile +``` + +### General Information + +By default, each job receives 30 minutes of *wall time*, or clock time. If you know that your job will not need more than a certain amount of time to run, request less than the maximum wall time, as this may allow your job to run sooner. To request 1 hour and 30 minutes of wall time: + +``` +$ sbatch -t 01:30:00 -N=1 -n=1 -p=cpu -A=accountname -q=standby myjobsubmissionfile +``` + +The `--nodes=` or `-N` value indicates how many compute nodes you would like for your job, and `--ntasks=` or `-n` value indicates the number of tasks you want to run. + +In some cases, you may want to request multiple nodes. To utilize multiple nodes, you will need to have a program or code that is specifically programmed to use multiple nodes such as with MPI. Simply requesting more nodes will not make your work go faster. Your code must support this ability. + +To request 2 compute nodes: + +``` +$ sbatch -t 01:30:00 -N=2 -n=16 -p=cpu -A=accountname -q=standby myjobsubmissionfile +``` + +By default, jobs on Negishi will share nodes with other jobs. + +If more convenient, you may also specify any command line options to `sbatch` from within your job submission file, using a special form of comment: + +``` +#!/bin/sh -l +# FILENAME: myjobsubmissionfile + +#SBATCH --account=accountname +#SBATCH --nodes=1 +#SBATCH --ntasks=1 +#SBATCH --partition=cpu +#SBATCH --qos=normal +#SBATCH --time=1:30:00 +#SBATCH --job-name myjobname + +# Print the hostname of the compute node on which this job is running. +/bin/hostname +``` + +If an option is present in both your job submission file and on the command line, the option on the command line will take precedence. + +After you submit your job with `SBATCH`, it may wait in queue for minutes, hours, or even weeks. How long it takes for a job to start depends on the specific queue, the resources and time requested, and other jobs already waiting in that queue requested as well. It is impossible to say for sure when any given job will start. For best results, request no more resources than your job requires. + +Once your job is submitted, you can [monitor the job status](../monitoring_job), wait for the job to complete, and [check the job output](../checking_output). + +​[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/vasp.md b/docs/userguides/negishi/run_jobs/vasp.md new file mode 100644 index 00000000..0453f098 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/vasp.md @@ -0,0 +1,58 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + + +# VASP + +The Vienna Ab initio Simulation Package (VASP) is a computer program for atomic scale materials modelling, e.g. electronic structure calculations and quantum-mechanical molecular dynamics, from first principles. + +## VASP License + +The VASP team allows only registered users who have purchased their own license to use the software and access is only given to the VASP release which is covered by the license of the respective research group. For those who are interested to use VASP on Negishi, please [contact support](/contact) to request access and provide your email address associated with your license for our verification. Once confirmed, the approved users will be given access to the `vasp5` or/and `vasp6` unix groups. + +Prospective users can use the command below to check their unix groups on the system. + +``` +$ id $USER +``` + +If you are interested to purchase and get a VASP license, please visit [VASP](http://vasp.at) website for more information. + +## VASP 5 and VASP 6 Installations + +Negishi provides **VASP 5.4.4.pl2** and **VASP 6.4.1** installations and modulefiles with our default environment compiler `gcc/12.2.0` and mpi library `openmpi/4.1.4`. Note that only license-approved users can load the VASP modulefile as below. + +You can use the VASP 5.4.4.pl2 module by: + +``` +$ module load vasp/5.4.4.pl2 +``` + +You can use the VASP 6.4.1 module by: + +``` +$ module load vasp/6.4.1 +``` + +Once a VASP module is loaded, you can choose one of the VASP executables to run your code: `vasp_std`, `vasp_gam`, and `vasp_ncl`. + +The VASP pseudopotential files are not provided on Negishi, you may need to bring your own POTCAR files. + +## Build your own VASP 5 and VASP 6 + +If you would like to use your own VASP on Negishi, please follow the instructions for [Installing VASP.6.X.X](https://www.vasp.at/wiki/index.php/Installing_VASP.6.X.X) and [Installing VASP.5.X.X](https://www.vasp.at/wiki/index.php/Installing_VASP.5.X.X). + +In the following sections, we provide some instructions about how to install VASP 5 and VASP 6 as well as bash job submit script on Negishi: + +* [VASP Job Submit Script](vasp/vasp-job-submit-script.md) +* [Build your own VASP 5](vasp/build-your-own-vasp-5.md) +* [Build your own VASP 6](vasp/build-your-own-vasp-6.md) + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/vasp/build-your-own-vasp-5.md b/docs/userguides/negishi/run_jobs/vasp/build-your-own-vasp-5.md new file mode 100644 index 00000000..128a260e --- /dev/null +++ b/docs/userguides/negishi/run_jobs/vasp/build-your-own-vasp-5.md @@ -0,0 +1,95 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Build your own VASP 5 + +For VASP 5.X.X version, VASP provide several templates of `makefile.include` in the `/arch` folder, which contain information such as precompiler options, compiler options, and how to link libraries. You can pick up one based on your system and preferred features. Here we provide some examples about how to install the `vasp.5.4.4.pl2.tgz` version on Negishi with different module environments. + +### Step 1: Download + +As a license holder, you can download the source code of VASP from the [VASP Portal](https://www.vasp.at/sign_in/portal/), we will not check your license in this case. + +Copy the VASP resource file `vasp.5.4.4.pl2.tgz` to the desired location, and unzip the file `tar zxvf vasp.5.4.4.pl2.tgz` to obtain the folder `/path/to/vasp-build-folder/vasp.5.4.4.pl2`and reveal its content. + +### Step 2: Prepare makefile.include + +We recommend to use GNU compilers parallelized using OpenMPI, combined with MKL for VASP compilation on Negishi. + +We are using MKL library, which include BLAS, LAPACK, ScaLAPACK, and FFTW as suggested at [VASP wiki](https://www.vasp.at/wiki/index.php/Installing_VASP.5.X.X#Requirements), and we can modify `makefile.include.linux_gnu` from `/arch` folder: + +``` +$ cd /path/to/vasp-build-folder/vasp.5.4.4.pl2 +cp arch/makefile.include.linux_gnu makefile.include +``` + +Here are the suggested changes for `makefile.include`, replace the lines between `DEBUG=-O0` and `OBJECTS= fftmpiw.o fftmpi_map.o fftw3d.o fft3dlib.o` with + +``` +# Intel MKL (FFTW, BLAS, LAPACK, and scaLAPACK) +LLIBS += -L${MKLROOT}/lib/intel64 -Wl,--no-as-needed -lmkl_gf_lp64 -lmkl_gnu_thread -lmkl_core -lmkl_scalapack_lp64 -lmkl_blacs_openmpi_lp64 -lgomp -lpthread -lm -ldl +INCS = -I$(MKLROOT)/include/fftw +FFLAGS += -march=znver3 + +# For gcc-10 and higher (comment out for older versions) +FFLAGS += -fallow-argument-mismatch +``` + +Remove all the `GPU stuff` at the end of `makefile.include` file + +Load the required modules: + +``` +module --force purge +module load gcc/12.2.0 openmpi/4.1.4 +module load intel-mkl/2019.9.304 +``` + +### Step 3: Make + +Build VASP with command `make all` to install all three executables `vasp_std`, `vasp_gam`, and `vasp_ncl` or use `make std` to install only the `vasp_std` executable. Use `make veryclean` to remove the build folder if you would like to start over the installation process. + +### Step 4: Test + +You can open an [Interactive session](../../run_jobs/interactive_jobs.md) to test the installed VASP with GNU/openMPI compilation, you may bring your own VASP test files: + +``` +$ cd /path/to/vasp-test-folder/ +module --force purge +module load gcc/12.2.0 openmpi/4.1.4 intel-mkl/2019.9.304 +module list +mpirun /path/to/vasp-build-folder/vasp.5.4.4.pl2/bin/vasp_std +``` + +### Step 5: submit a bash job + +To submit a bash job with your own compiled VASP on Negishi, here is an example about how to set up your environment and launch MPI code. + +``` +#!/bin/bash + +#SBATCH -A myqueuename # Queue name(use 'slist' command to find queues' name) +#SBATCH --nodes=1 # Total # of nodes +#SBATCH --ntasks=64 # Total # of MPI tasks +#SBATCH --time=1:00:00 # Total run time limit (hh:mm:ss) +#SBATCH -J myjobname # Job name +#SBATCH -o myjob.o%j # Name of stdout output file +#SBATCH -e myjob.e%j # Name of stderr error file + +# Manage processing environment,load compilers and applications. +module purge +module load gcc/12.2.0 openmpi/4.1.4 intel-mkl/2019.9.304 +module list +export PATH=/path/to/vasp-build-folder/vasp.x.x.x/bin:$PATH + +# Launch MPI code +mpirun -np $SLURM_NTASKS vasp_std +``` + +[**Back to the VASP section**](../vasp.md) diff --git a/docs/userguides/negishi/run_jobs/vasp/build-your-own-vasp-6.md b/docs/userguides/negishi/run_jobs/vasp/build-your-own-vasp-6.md new file mode 100644 index 00000000..3a8f44cd --- /dev/null +++ b/docs/userguides/negishi/run_jobs/vasp/build-your-own-vasp-6.md @@ -0,0 +1,97 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Build your own VASP 6 + +For VASP 6.X.X version, VASP provide several templates of [`makefile.include`](https://www.vasp.at/wiki/index.php/Makefile.include), which contain information such as precompiler options, compiler options, and how to link libraries. You can pick up one based on your system and preferred features . Here we provide some examples about how to install `vasp 6.4.1` on Negishi with different module environments. + +### Step 1: Download + +As a license holder, you can download the source code of VASP from the [VASP Portal](https://www.vasp.at/sign_in/portal/), we will not check your license in this case. + +Copy the VASP resource file `vasp.6.4.1.tgz` to the desired location, and unzip the file `tar zxvf vasp.6.4.1.tgz` to obtain the folder `/path/to/vasp-build-folder/vasp.6.4.1` and reveal its content. + +### Step 2: Prepare makefile.include + +We recommend to use GNU compilers parallelized using OpenMPI + OpenMP, combined with MKL for VASP build on Negishi. + +We are using MKL library, which include BLAS, LAPACK, ScaLAPACK, and FFTW as suggested at [VASP wiki](https://www.vasp.at/wiki/index.php/Installing_VASP.6.X.X#Requirements). We can modify `makefile.include.gnu_ompi_mkl_omp` from `/arch` folder to fit for system setup: + +``` +$ cd /path/to/vasp-build-folder/vasp.6.4.1 +$ cp makefile.include.gnu_ompi_mkl_omp makefile.include +``` + +Here are the suggested changes for `makefile.include`: + +* change `VASP_TARGET_CPU ?= -march=native` to + + ``` + VASP_TARGET_CPU ?= -march=znver3 + ``` +* remove `MKLROOT ?= /path/to/your/mkl/installation` +* change `LLIBS_MKL =` to + + ``` + LLIBS += + ``` +* comment out or remove all the lines after `INCS = -I$(MKLROOT)/include/fftw` + +Then, load the required modules: + +``` +$ module purge +$ module load gcc/12.2.0 openmpi/4.1.4 +$ module load intel-mkl/2019.9.304 +``` + +### Step 3: Make + +Open `makefile`, make sure the first line is `VERSIONS = std gam ncl`. + +Build VASP with command `make all` to install all three executables `vasp_std`, `vasp_gam`, and `vasp_ncl` or use `make std` to install only the `vasp_std` executable. Use `make veryclean` to remove the build folder if you would like to start over the installation process. + +### Step 4: Test + +You can open an [Interactive session](../../run_jobs/interactive_jobs.md) to test the installed VASP 6. Here is an example of testing above installed VASP 6.4.1 with GNU compilers and OpenMPI: + +``` +$ cd /path/to/vasp-build-folder/vasp.6.4.1/testsuite +$ module purge +$ module load gcc/12.2.0 openmpi/4.1.4 intel-mkl/2019.9.304 +$ ./runtest +``` + +### Step 5: Submit a bash job + +To submit a bash job with your own compiled VASP on Negishi, here is an example about how to set up your environment and launch MPI code. + +``` +#!/bin/bash + +#SBATCH -A myqueuename # Queue name(use 'slist' command to find queues' name) +#SBATCH --nodes=1 # Total # of nodes +#SBATCH --ntasks=64 # Total # of MPI tasks +#SBATCH --time=1:00:00 # Total run time limit (hh:mm:ss) +#SBATCH -J myjobname # Job name +#SBATCH -o myjob.o%j # Name of stdout output file +#SBATCH -e myjob.e%j # Name of stderr error file + +# Manage processing environment,load compilers and applications. +module purge +module load gcc/12.2.0 openmpi/4.1.4 intel-mkl/2019.9.304 +module list +export PATH=/path/to/vasp-build-folder/vasp.x.x.x/bin:$PATH + +# Launch MPI code +mpirun -np $SLURM_NTASKS vasp_std +``` + +[**Back to the VASP section**](../vasp.md) diff --git a/docs/userguides/negishi/run_jobs/vasp/vasp-job-submit-script.md b/docs/userguides/negishi/run_jobs/vasp/vasp-job-submit-script.md new file mode 100644 index 00000000..867e0c2c --- /dev/null +++ b/docs/userguides/negishi/run_jobs/vasp/vasp-job-submit-script.md @@ -0,0 +1,34 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# VASP Job Submit Script + +This shows an example of a job submission file for running VASP pre-built on Negishi: + +``` +#!/bin/bash + +#SBATCH -A myqueuename # Queue name(use 'slist' command to find queues' name) +#SBATCH --nodes=1 # Total # of nodes +#SBATCH --ntasks=64 # Total # of MPI tasks +#SBATCH --time=1:00:00 # Total run time limit (hh:mm:ss) +#SBATCH -J myjobname # Job name +#SBATCH -o myjob.o%j # Name of stdout output file +#SBATCH -e myjob.e%j # Name of stderr error file + +# Manage processing environment, load compilers and applications. +module load vasp/5.4.4.pl2 # or module load vasp/6.4.1 +module list + +# Launch MPI code +mpirun -np $SLURM_NTASKS vasp_std +``` + +[**Back to the VASP section**](../vasp.md) diff --git a/docs/userguides/negishi/run_jobs/windows.md b/docs/userguides/negishi/run_jobs/windows.md new file mode 100644 index 00000000..4bdb8f09 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/windows.md @@ -0,0 +1,42 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Windows + +Windows virtual machines (VMs) are supported as batch jobs on HPC systems. This section illustrates how to submit a job and run a Windows instance in order to run Windows applications on the high-performance computing systems. + +The following images are pre-configured and made available by staff: + +* Windows 2016 Server Basic (minimal software pre-loaded) +* Windows 2016 Server GIS (GIS Software Stack pre-loaded) + +The Windows VMs can be launched in two fashions: + +* [Menu Launcher](windows/launcher.md) - Point and click to start +* [Command Line](windows/cmd.md) - Advanced and customized usage + +Click each of the above links for detailed instructions on using them. + +## Software Provided in Pre-configured Virtual Machines + +The Windows 2016 Base server image available on ${resource.name} has the following software packages preloaded: + +* Anaconda Python 2 and Python 3 +* JMP 13 +* Matlab R2017b +* Microsoft Office 2016 +* Notepad++ +* NVivo 12 +* Rstudio +* Stata SE 15 +* VLC Media Player + + +[**Back to the Running Jobs section**](index.md) diff --git a/docs/userguides/negishi/run_jobs/windows/cmd.md b/docs/userguides/negishi/run_jobs/windows/cmd.md new file mode 100644 index 00000000..36909d97 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/windows/cmd.md @@ -0,0 +1,57 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Command line + +If you wish to work with Windows VMs on the command line or work into scripted workflows you can interact directly with the Windows system: + +Copy a Windows 2016 Server VM image to your storage. Scratch or Research Data Depot are good locations to save a VM image. If you are using scratch, remember that **scratch spaces are temporary**, and be sure to safely back up your disk image somewhere permanent, such as Research Data Depot or Fortress. To copy a basic image: + +``` +$ cp /apps/external/apps/windows/images/latest.qcow2  $RCAC_SCRATCH/windows.qcow2 +``` + +To copy a GIS image: + +``` +$ cp /depot/itap/windows/gis/2k16.qcow2 $RCAC_SCRATCH/windows.qcow2 +``` + +To launch a virtual machine in a batch job, use the "windows" script, specifying the path to your Windows virtual machine image. With no other command-line arguments, the `windows` script will autodetect a number cores and memory for the Windows VM. A Windows network connection will be made to your home directory. To launch: + +``` +$ windows -i $RCAC_SCRATCH/windows.qcow2 +``` + +## Command line options: + +``` +-i (For example, $RCAC_SCRATCH/windows-2k16.qcow2) +-m G (For example, 32G) +-c (For example, 20) +-s (UNIX Path to map as a drive, for example, $RCAC_SCRATCH) +-b (If present, launches VM in background. Use VNC to connect to Windows.) +``` + +To launch a virtual machine with 32GB of RAM, 20 cores, and a network mapping to your home directory: + +``` +$ windows -i /path/to/image.qcow2 -m 32G -c 20 -s $HOME +``` + +To launch a virtual machine with 16GB of RAM, 10 cores, and a network mapping to your Data Depot space: + +``` +$ windows -i /path/to/image.qcow2 -m 16G -c 10 -s /depot/mylab +``` +The Windows 2016 server desktop will open, and automatically log in as an administrator, so that you can install any software into the Windows virtual machine that your research requires. Changes to the image will be stored in the file specified with the `-i` option. + + +[**Back to the Windows section**](../windows.md) diff --git a/docs/userguides/negishi/run_jobs/windows/launcher.md b/docs/userguides/negishi/run_jobs/windows/launcher.md new file mode 100644 index 00000000..2c1b1396 --- /dev/null +++ b/docs/userguides/negishi/run_jobs/windows/launcher.md @@ -0,0 +1,39 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Menu Launcher + +Windows VMs can be easily launched through the [thinlinc](../../../accounts/#thinlinc) remote desktop environment. + +* Log in via [thinlinc](../../../accounts/#thinlinc). +* Click on Applications menu in the upper left corner. +* Look under the Cluster Software menu. +* The "Windows 10" launcher will launch a VM directly on the front-end. +* Follow the dialogs to set up your VM. + +![Thinlinc Applications list](../../../../../assets/images/userguides/examples/windows.png) + +*Find Windows 10 under the 'Cluster Software' option in the list of Applications.* + +The dialog menus will walk you through setting up and loading your VM. + +* You can choose to create a new image or load a saved image. +* New VMs should be saved on Scratch or Research Data Depot as they are too large for Home Directories. +* If you are using scratch, remember that **scratch spaces are temporary**, and be sure to safely back up your disk image somewhere permanent, such as Research Data Depot or Fortress. + +You will also be prompted to select a storage space to mount on your image (Home, Scratch, or Data Depot). You can only choose one to be mounted. It will appear on a shortcut on the desktop once the VM loads. + +## Notes + +Using the menu launcher will launch automatically select reasonable CPU and memory values. If you wish to choose other options or work Windows VMs into scripted workflows see the section on [using the command line](cmd.md). + + + +[**Back to the Windows section**](../windows.md) diff --git a/docs/userguides/negishi/software.md b/docs/userguides/negishi/software.md new file mode 100644 index 00000000..538c8a4f --- /dev/null +++ b/docs/userguides/negishi/software.md @@ -0,0 +1,88 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{% set resource = "negishi" %} + +# Software on Negishi + +## Software Catalog + +A comprehensive list of centrally installed software applications can be found here: + +[Software Catalog](../../software/app_catalog.md) + +## Module system + +{{ module_system(resource) }} + +## Running the Apps +### Find available apps in the terminal +In addition to searching the software catalog for available applications, one can generate a list via the terminal: + +``` bash +$ module avail +|---------------------- Core Applications --------------------- + amduprof/3.4-502 hyper-shell/2.1.0 openblas/0.3.21 + anaconda/2021.05-py38 hyper-shell/2.4.0 openjdk/1.8.0_265-b01 + anaconda/2022.10-py39 hyper-shell/2.5.1 openjdk/11.0.17_8 + anaconda/2024.02-py311 (D) hyper-shell/2.5.2 ovito/3.11.0 +[MORE...] +``` +### View module prequisites and license information +After finding the module that you want to load, use 'module spider' to find any prerequisites or license information, if applicable: + +``` bash +$ module spider hypershell + +------------------------------------------------------------- + hypershell: +------------------------------------------------------------- + Description: + A cross-platform, high-throughput computing utility for processing shell commands over a + distributed, asynchronous queue. + + Versions: + hypershell/2.6.2 + hypershell/2.6.5 + hypershell/2.7.0 + +``` +### Load the module +Use the command specified in the 'module spider' output to load your software module: + +``` bash +module load hypershell/2.7.0 +``` + +### Running GUI versions of apps +If the app you want to use has a GUI, you can also login to {{ resource }} via Thinlinc. More information on this process can be found [here](accounts.md#thinlinc). + +## ROCm Containers + +Negishi's GPU sub-cluster (Sub-cluster G) is equipped with AMD MI210 GPUs. A selection of GPU-enabled ROCm application containers from the AMD Infinity Hub collection is installed. + +Users can download additional ROCm containers from the [AMD Infinity Hub](https://www.amd.com/en/technologies/infinity-hub/) and run them directly using Apptainer/Singularity. A subset of pre-downloaded ROCm containers wrapped into convenient software modules are also provided. + +To see the lists of ROCm containers available as modules, use: + +```bash +$ module avail rocm +``` + +More information on pre-downloaded ROCm containers can be found [here](./../../../software/rocm_catalog). + +## BioContainers + +Pre-downloaded bioinformatics containers with module wrappers are available. You can load them as standard modules: + +```bash +$ module load biocontainers +``` +More information on pre-downloaded ROCm containers can be found [here](https://biocontainer-doc.readthedocs.io/). diff --git a/docs/userguides/negishi/storage.md b/docs/userguides/negishi/storage.md new file mode 100644 index 00000000..61849391 --- /dev/null +++ b/docs/userguides/negishi/storage.md @@ -0,0 +1,47 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar + - remender +resource: Negishi +search: + boost: 2 +--- + +# File Storage and Transfer + +Learn more about file storage transfer for Negishi. + +## Storage Options + +File storage options on RCAC systems include long-term storage (home directories, depot, Fortress) and short-term storage (scratch directories, /tmp directory). Each option has different performance and intended uses, and some options vary from system to system as well. Daily snapshots of home directories are provided for a limited time for accidental deletion recovery. Scratch directories and temporary storage are not backed up and old files are regularly purged from scratch and /tmp directories. More details about each storage option appear below. + +- [Home Directory](storage/home_directory.md) +- [Scratch Space](storage/scratch_space.md) +- [/tmp Directory](storage/tmp_directory.md) +- [Long-Term Storage](storage/long_term_storage.md) + +### Other Storage Topics +- [Storage Quota / Limits](storage/storage_quota.md) +- [Storage Environment Variables](storage/environment_variables.md) +- [Archive and Compression](storage/archive_and_compression.md) +- [Sharing](storage/sharing.md) + +## File Transfer + +Negishi supports several methods for file transfer. Use the links below to learn more about these methods. + +- [Globus](storage/globus.md) +- [Windows Network Drive / SMB](storage/windows_network_drive.md) +- [SCP](storage/scp.md) +- [FTP / SFTP](storage/ftp_sftp.md) +- [HSI](storage/hsi.md) +- [HTAR](storage/htar.md) +- [Copying files from Purdue IT research computing home directory to Negishi](storage/copyhome.md) +## Lost File Recovery + +- [Lost File Recovery](storage/recover.md) + +[**Back to Negishi User Guide**](./index.md) diff --git a/docs/userguides/negishi/storage/archive_and_compression.md b/docs/userguides/negishi/storage/archive_and_compression.md new file mode 100644 index 00000000..4d9934a2 --- /dev/null +++ b/docs/userguides/negishi/storage/archive_and_compression.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/archive_and_compression.md" + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/copyhome.md b/docs/userguides/negishi/storage/copyhome.md new file mode 100755 index 00000000..6fc85319 --- /dev/null +++ b/docs/userguides/negishi/storage/copyhome.md @@ -0,0 +1,57 @@ +--- +tags: + - Negishi +authors: + - hkashgar +resource: Negishi +search: + boost: 2 +--- +# Copying files from Purdue IT research computing home directory to Negishi + +The Negishi home directory and its contents are specific to the Negishi cluster, and are not available on other RCAC machines. For people having access to other Community Clusters and Negishi, *there is no automatic copying or synchronization between main and Negishi home directories*. At your discretion, you can manually copy all or parts of your main research computing home to Negishi using one of the methods described below. + +Please note that copying may fail if the size of your research computing home directory is larger than the Negishi one's quota. Please [check](../storage_quota) usage and limits before proceeding! + +### Complete copy + +For your convenience, a custom tool `copy-rcac-home` is provided to simplify at-will duplication of your main research computing home directory into Negishi. The tool performs a complete 1-to-1 copy using `rsync -auH` (with exception of a narrow subset of system-specific service files). + +To use the tool, simply type `copy-rcac-home` in a terminal window on a Negishi front-end or compute node: + +``` + +$ copy-rcac-home + + This script will copy entire contents of your main RCAC + home directory into your Negishi cluster's $HOME. + + Note: copying may fail if the size of your RCAC home directory + is larger than your quota on the Negishi one (25GB). + BEFORE PROCEEDING, please run 'myquota' command on another + cluster to see your usage there and judge whether it would fit! + +Would you like to proceed? [Y/n]: +``` + +At this stage answering `yes` will proceed with copying, or you can respond with a `no` (or `Ctrl-C`) to cancel. See `copy-rcac-home --help` for more details on the tool. + +### Partial copy + +Desired parts (or whole) of your research computing home directories can be copied to Negishi via any of the home directories' supported [transfer methods](../#file-transfer), such as SCP, SFTP, rsync, or Globus. + +* **Example:** recursive copying of a subdirectory from RCAC home directory into Negishi home using `scp`. + + ``` + + (if you are on Negishi, use other cluster name for the remote part) + $ scp -pr myothercluster.rcac.purdue.edu:somedirectory/ ~/ + + (if you are on another cluster, use Negishi for the remote part) + $ scp -pr somedirectory/ myusername@negishi.rcac.purdue.edu:~/ + ``` +* **Example:** copying using Globus. + + Search collections for *"Purdue Research Computing - Home Directories"* and *"Purdue Negishi Cluster"* endpoints, respectively, then transfer desired files and/or directories as usual. + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/environment_variables.md b/docs/userguides/negishi/storage/environment_variables.md new file mode 100644 index 00000000..f3a4b060 --- /dev/null +++ b/docs/userguides/negishi/storage/environment_variables.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ environment_variables(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/ftp_sftp.md b/docs/userguides/negishi/storage/ftp_sftp.md new file mode 100644 index 00000000..082048fd --- /dev/null +++ b/docs/userguides/negishi/storage/ftp_sftp.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ ftp_sftp_snippet(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/globus.md b/docs/userguides/negishi/storage/globus.md new file mode 100644 index 00000000..2cabca11 --- /dev/null +++ b/docs/userguides/negishi/storage/globus.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ globus_snippet(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/home_directory.md b/docs/userguides/negishi/storage/home_directory.md new file mode 100644 index 00000000..606d94c1 --- /dev/null +++ b/docs/userguides/negishi/storage/home_directory.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/home_directory.md" + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/hsi.md b/docs/userguides/negishi/storage/hsi.md new file mode 100644 index 00000000..7a2b6858 --- /dev/null +++ b/docs/userguides/negishi/storage/hsi.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ hsi_snippet(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/htar.md b/docs/userguides/negishi/storage/htar.md new file mode 100644 index 00000000..c5b720a9 --- /dev/null +++ b/docs/userguides/negishi/storage/htar.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ htar_snippet(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/long_term_storage.md b/docs/userguides/negishi/storage/long_term_storage.md new file mode 100644 index 00000000..34b4d4ac --- /dev/null +++ b/docs/userguides/negishi/storage/long_term_storage.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - hkashgar + - jin456 +resource: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/long_term_storage.md" + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/recover.md b/docs/userguides/negishi/storage/recover.md new file mode 100644 index 00000000..6e4ee59c --- /dev/null +++ b/docs/userguides/negishi/storage/recover.md @@ -0,0 +1,36 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +# Lost File Recovery + +Negishi is protected against accidental file deletion through a series of snapshots taken every night just after midnight. Each snapshot provides the state of your files at the time the snapshot was taken. It does so by storing only the files which have changed between snapshots. A file that has not changed between snapshots is only stored once but will appear in every snapshot. This is an efficient method of providing snapshots because the snapshot system does not have to store multiple copies of every file. + +These snapshots are kept for a limited time at various intervals. RCAC keeps nightly snapshots for 7 days, weekly snapshots for 4 weeks, and monthly snapshots for 3 months. This means you will find snapshots from the last 7 nights, the last 4 Sundays, and the last 3 first of the months. Files are available going back between two and three months, depending on how long ago the last first of the month was. Snapshots beyond this are not kept. + +**Only files which have been saved during an overnight snapshot are recoverable.** If you lose a file the same day you created it, the file is **not** recoverable because the snapshot system has not had a chance to save the file. + +**Snapshots are not a substitute for regular backups.** It is the responsibility of the researchers to back up any important data to the Fortress Archive. Negishi **does** protect against hardware failures or physical disasters through other means however these other means are also **not** substitutes for backups. + +Files in scratch directories are not recoverable. Files in scratch directories are not backed up. If you accidentally delete a file, a disk crashes, or old files are purged, they cannot be restored. + +Negishi offers several ways for researchers to access snapshots of their files. + + +* [flost](recover/flost.md) + +* [Mac OS X](recover/mac.md) + +* [Windows](recover/windows.md) + +* [Manual Browsing](recover/manual.md) + + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/recover/flost.md b/docs/userguides/negishi/storage/recover/flost.md new file mode 100644 index 00000000..6e3c6ee1 --- /dev/null +++ b/docs/userguides/negishi/storage/recover/flost.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ recover_flost_snippet(resource) }} + +[**Back to the Recovery section**](../recover.md) diff --git a/docs/userguides/negishi/storage/recover/mac.md b/docs/userguides/negishi/storage/recover/mac.md new file mode 100644 index 00000000..f004f36c --- /dev/null +++ b/docs/userguides/negishi/storage/recover/mac.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ recover_mac_snippet(resource) }} + +[**Back to the Recovery section**](../recover.md) diff --git a/docs/userguides/negishi/storage/recover/manual.md b/docs/userguides/negishi/storage/recover/manual.md new file mode 100644 index 00000000..b2f5ff4a --- /dev/null +++ b/docs/userguides/negishi/storage/recover/manual.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ recover_manual_snippet(resource) }} + +[**Back to the Recovery section**](../recover.md) diff --git a/docs/userguides/negishi/storage/recover/windows.md b/docs/userguides/negishi/storage/recover/windows.md new file mode 100644 index 00000000..708a869e --- /dev/null +++ b/docs/userguides/negishi/storage/recover/windows.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ recover_windows_snippet(resource) }} + +[**Back to the Recovery section**](../recover.md) diff --git a/docs/userguides/negishi/storage/scp.md b/docs/userguides/negishi/storage/scp.md new file mode 100644 index 00000000..e9924222 --- /dev/null +++ b/docs/userguides/negishi/storage/scp.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ scp_file_snippet(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/scratch_space.md b/docs/userguides/negishi/storage/scratch_space.md new file mode 100644 index 00000000..b6b2006b --- /dev/null +++ b/docs/userguides/negishi/storage/scratch_space.md @@ -0,0 +1,15 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar + +resource: Negishi +search: + boost: 2 +--- + +{{ scratch_space(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/sharing.md b/docs/userguides/negishi/storage/sharing.md new file mode 100644 index 00000000..c1340d32 --- /dev/null +++ b/docs/userguides/negishi/storage/sharing.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ sharing_snippet(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/storage_quota.md b/docs/userguides/negishi/storage/storage_quota.md new file mode 100644 index 00000000..997ffe7a --- /dev/null +++ b/docs/userguides/negishi/storage/storage_quota.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ storage_quota(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/tmp_directory.md b/docs/userguides/negishi/storage/tmp_directory.md new file mode 100644 index 00000000..32a024be --- /dev/null +++ b/docs/userguides/negishi/storage/tmp_directory.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +--8<-- "docs/snippets/tmp_directory.md" + +[**Back to the Storage section**](../storage.md) diff --git a/docs/userguides/negishi/storage/windows_network_drive.md b/docs/userguides/negishi/storage/windows_network_drive.md new file mode 100644 index 00000000..7659254b --- /dev/null +++ b/docs/userguides/negishi/storage/windows_network_drive.md @@ -0,0 +1,14 @@ +--- +tags: + - Negishi +authors: + - jin456 + - hkashgar +resource: Negishi +search: + boost: 2 +--- + +{{ windows_network_drive(resource) }} + +[**Back to the Storage section**](../storage.md) diff --git a/mkdocs.yml b/mkdocs.yml index eff519a9..4ec8fc5c 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -154,6 +154,17 @@ nav: - Gateway (Open OnDemand): userguides/gilbreth/gateway.md - Compiling Source Code: userguides/gilbreth/compile.md - Frequently Asked Questions: userguides/gilbreth/faqs.md + - Negishi: + - userguides/negishi/index.md + - Negishi Overview: userguides/negishi/overview.md + - Biography of Negishi: userguides/negishi/biography.md + - Accounts: userguides/negishi/accounts.md + - Software: userguides/negishi/software.md + - Running Jobs: userguides/negishi/run_jobs/index.md + - File Storage and Transfer: userguides/negishi/storage.md + - Gateway (Open OnDemand): userguides/negishi/gateway.md + - Compiling Source Code: userguides/negishi/compile.md + - Frequently Asked Questions: userguides/negishi/faqs.md - Scholar: - userguides/scholar/index.md - Scholar Overview: userguides/scholar/overview.md @@ -180,7 +191,6 @@ nav: - Web Server: userguides/geddes/examples/webserver.md - R Shiny: userguides/geddes/examples/r-shiny.md - Troubleshooting: userguides/geddes/troubleshooting.md - - Negishi: https://www.rcac.purdue.edu/knowledge/negishi - Hammer: https://www.rcac.purdue.edu/knowledge/hammer - Rossmann: https://www.rcac.purdue.edu/knowledge/rossmann - Weber: https://www.rcac.purdue.edu/knowledge/weber