From bfe6ebd39c0d33aacd7a661d73ae6f8b52be8504 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 30 Jun 2026 12:31:17 +0800 Subject: [PATCH 01/37] Add Harper workflow for grammar and style checks --- .github/workflows/harper.yml | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) create mode 100644 .github/workflows/harper.yml diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml new file mode 100644 index 0000000000..531049e92e --- /dev/null +++ b/.github/workflows/harper.yml @@ -0,0 +1,23 @@ +name: Harper + +on: + pull_request: + workflow_dispatch: + +jobs: + harper: + name: Grammar and Style Check + runs-on: ubuntu-latest + + steps: + - name: Checkout + uses: actions/checkout@v6 + + - name: Install Rust + uses: dtolnay/rust-toolchain@stable + + - name: Install Harper CLI + run: cargo install harper-cli --locked + + - name: Run Harper + run: harper-cli lint docs src/pages From 0d20fcf27204eeeda53d3f6d6e973cf37783f189 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 30 Jun 2026 12:41:40 +0800 Subject: [PATCH 02/37] Remove Rust installation step from workflow --- .github/workflows/harper.yml | 3 --- 1 file changed, 3 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 531049e92e..d5d3d43da4 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -13,9 +13,6 @@ jobs: - name: Checkout uses: actions/checkout@v6 - - name: Install Rust - uses: dtolnay/rust-toolchain@stable - - name: Install Harper CLI run: cargo install harper-cli --locked From 5b966a796ae7e3b660c9070391e5043fc6d19966 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 30 Jun 2026 12:47:48 +0800 Subject: [PATCH 03/37] Update Harper CLI installation method in workflow --- .github/workflows/harper.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index d5d3d43da4..6e51988c6c 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -14,7 +14,7 @@ jobs: uses: actions/checkout@v6 - name: Install Harper CLI - run: cargo install harper-cli --locked + run: cargo install --git https://github.com/Automattic/harper harper-cli --locked - name: Run Harper run: harper-cli lint docs src/pages From 56c9370b6e1020ce3987c648c56b7f94bd6a2897 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 30 Jun 2026 12:54:45 +0800 Subject: [PATCH 04/37] Update Harper CLI lint command to use find --- .github/workflows/harper.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 6e51988c6c..1da0178edc 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -17,4 +17,4 @@ jobs: run: cargo install --git https://github.com/Automattic/harper harper-cli --locked - name: Run Harper - run: harper-cli lint docs src/pages + run: find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | xargs -0 harper-cli lint From 1613f33ced35acc11d74f9eff15d42ac6c371377 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Thu, 2 Jul 2026 00:49:39 +0800 Subject: [PATCH 05/37] Add new terms to the harper dictionary --- .harper-dictionary.txt | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/.harper-dictionary.txt b/.harper-dictionary.txt index 79a397e136..2bf6d8bc9c 100644 --- a/.harper-dictionary.txt +++ b/.harper-dictionary.txt @@ -1 +1,9 @@ HPC +DDP +SPMD +DPP +GWDG +SageMaker +Slurm +SBATCH +py From cb8179bd1e930161edcd95d6464a03c0d4e0ed84 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Thu, 2 Jul 2026 16:17:29 +0800 Subject: [PATCH 06/37] Update .harper-dictionary.txt --- .harper-dictionary.txt | 149 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 149 insertions(+) diff --git a/.harper-dictionary.txt b/.harper-dictionary.txt index 2bf6d8bc9c..77eaec0b9a 100644 --- a/.harper-dictionary.txt +++ b/.harper-dictionary.txt @@ -7,3 +7,152 @@ SageMaker Slurm SBATCH py +TabItem +mdx +br +vpn +html +png +SLURM +GWDG +vLLM +vllm +Gemma +NetID +CUI +SRDE +srde +OSP +GOIS +gDTN +conda +QuickLinks +QuickLink +req +cvf +rclone +nvidia +smi +Nsight +ETL +ffcv +NVIDIA +DALI +ImageNet +linux +TMPDIR +HDF5 +hdf5 +LMDB +OS +np +lmd +lmdb +jpeg +Bcolz +Zarr +Globus +nyu +inode +r +julia +OOD +du +s +hr +VSCode +mv +ln +vscode +RStudio +rstudio +ide +myquota +mailto +rclone +RClone +Gb +GB +TB +Tb +MB +Mb +Descr +pr +XXXXX +br +srun +pty +sbatch +c4 +t2 +smi +NVIDIA +NVML +P0 +P12 +nvtop +sdiag +slurmctld +DBD +rpc +cnt +Backfiring +etc +gettimeofday +seff +rpsquota +sched +sacct +walltime +alphafold +renv +knitro +amd +lammps +matlab +mathematica +crystal17 +namd +squashfs +orca +gaussian +vnc +hadoop +sas +xvfb +jupyter +schrodinger +comsol +stata +Lmod +sif +sqf +apptainer +Guano +HuggingFace +rw +ModuleNotFoundError +ro +Xeon +H200 +SIF +L40S +OCI +Infiniband +WD40 +N/A +H100 +A100 +LINPACK +Tandon +SoE +Courant +CILVR +anaconda3 +ubuntu +whoami +pwd +ood +salloc +sinfo From 5b7af723dfce2a5552f5ec877d2a3757fa8671f9 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Thu, 2 Jul 2026 17:02:26 +0800 Subject: [PATCH 07/37] Update harper.yml --- .github/workflows/harper.yml | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 1da0178edc..29828e46fc 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -16,5 +16,13 @@ jobs: - name: Install Harper CLI run: cargo install --git https://github.com/Automattic/harper harper-cli --locked + - name: Show Harper CLI Help + run: | + harper-cli --help + echo "----------------" + harper-cli lint --help + - name: Run Harper - run: find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | xargs -0 harper-cli lint + run: | + find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ + xargs -0 harper-cli lint From 4118ded349589054d720b6aec61416a520b0391e Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Thu, 2 Jul 2026 17:12:46 +0800 Subject: [PATCH 08/37] Update harper.yml --- .github/workflows/harper.yml | 9 ++------- 1 file changed, 2 insertions(+), 7 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 29828e46fc..3e68ac5996 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -16,13 +16,8 @@ jobs: - name: Install Harper CLI run: cargo install --git https://github.com/Automattic/harper harper-cli --locked - - name: Show Harper CLI Help - run: | - harper-cli --help - echo "----------------" - harper-cli lint --help - - name: Run Harper run: | find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ - xargs -0 harper-cli lint + xargs -0 harper-cli lint \ + --user-dict-path ./.harper-dictionary.txt From e7c5300b841afeda596946e3586e3a7b92f74871 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 00:00:33 +0800 Subject: [PATCH 09/37] Update harper.yml --- .github/workflows/harper.yml | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 3e68ac5996..6995d021dc 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -16,6 +16,14 @@ jobs: - name: Install Harper CLI run: cargo install --git https://github.com/Automattic/harper harper-cli --locked + - name: Generate Harper spelling report + run: | + find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ + xargs -0 harper-cli lint \ + --user-dict-path ./.harper-dictionary.txt \ + --only Spelling::SpellCheck \ + --format compact || true + - name: Run Harper run: | find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ From ebb69ab03d4f5c7bde001fcfacdef80063a56845 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 00:36:52 +0800 Subject: [PATCH 10/37] Change report generation from spelling to compact --- .github/workflows/harper.yml | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 6995d021dc..9a6e517086 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -16,12 +16,11 @@ jobs: - name: Install Harper CLI run: cargo install --git https://github.com/Automattic/harper harper-cli --locked - - name: Generate Harper spelling report + - name: Generate Harper compact report run: | find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ xargs -0 harper-cli lint \ --user-dict-path ./.harper-dictionary.txt \ - --only Spelling::SpellCheck \ --format compact || true - name: Run Harper From 181ec9945576a213ab5bb625b7db4a498790b6c4 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 01:26:12 +0800 Subject: [PATCH 11/37] Update harper.yml --- .github/workflows/harper.yml | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 9a6e517086..4a98516f7f 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -16,12 +16,16 @@ jobs: - name: Install Harper CLI run: cargo install --git https://github.com/Automattic/harper harper-cli --locked - - name: Generate Harper compact report + - name: Generate Harper JSON Report run: | find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ xargs -0 harper-cli lint \ --user-dict-path ./.harper-dictionary.txt \ - --format compact || true + --format json \ + > harper-report.json || true + + - name: Show Harper Report + run: cat harper-report.json - name: Run Harper run: | From 8d39bb588b87c2d1cb8fae64d0fdb569915d93fe Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 10:21:47 +0800 Subject: [PATCH 12/37] make harper not fail --- .github/workflows/harper.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 4a98516f7f..41501155a0 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -31,4 +31,4 @@ jobs: run: | find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ xargs -0 harper-cli lint \ - --user-dict-path ./.harper-dictionary.txt + --user-dict-path ./.harper-dictionary.txt || true From 1d83d13e56e4dd17514dbc40d576b73c84e4cde1 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 11:01:16 +0800 Subject: [PATCH 13/37] Update harper.yml --- .github/workflows/harper.yml | 24 +++++++++++++++++++++--- 1 file changed, 21 insertions(+), 3 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 41501155a0..d9d197a457 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -24,11 +24,29 @@ jobs: --format json \ > harper-report.json || true - - name: Show Harper Report - run: cat harper-report.json + - name: Extract Spelling Candidates + run: | + python - <<'PY' + import json + + with open("harper-report.json", encoding="utf-8") as f: + reports = json.load(f) + + words = set() + for report in reports: + for lint in report.get("lints", []): + if lint.get("kind") == "Spelling": + word = lint.get("matched_text") + if word: + words.add(word) + + print("Suggested dictionary candidates:") + for word in sorted(words, key=str.lower): + print(word) + PY - name: Run Harper run: | find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ xargs -0 harper-cli lint \ - --user-dict-path ./.harper-dictionary.txt || true + --user-dict-path ./.harper-dictionary.txt || true From e823000e3465e74a38cec2761a2655fae8ab02e6 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 12:37:00 +0800 Subject: [PATCH 14/37] Update .harper-dictionary.txt --- .harper-dictionary.txt | 524 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 524 insertions(+) diff --git a/.harper-dictionary.txt b/.harper-dictionary.txt index 77eaec0b9a..f4f6fe2f53 100644 --- a/.harper-dictionary.txt +++ b/.harper-dictionary.txt @@ -156,3 +156,527 @@ pwd ood salloc sinfo +Act5C +admixtools +advanpix +ai +AIfSR +AllocCPUS +allowfullscreen +amanda +amdahl +AMPLab +amplgurobilink +AnyConnect +api +aql +argparse +ArgumentParser +argv +Aria2 +aria2 +AutoModel +Autoscaling +B. +B2 +bai +baiUrlTemplate +BAM +bcftools +BDGP5 +bgzip +BigBed +BigTable +bioconda +bioinformatic +Bourne +Burwood +bw +c1 +c5 +c9b4b +CACHEDIR +CANCELLED +cancelled +Cancelling +cancelling +Carahsoft +catalogue +Catalogue +cd +CentOS +centos +Ceph +cftuvSUX +cgsb +chmod +chrom +chromStart +CloudBank +CLS +cmd +Colab +ColdFront +cos6 +cp38 +cpuinfo +cpus +CRAN +cs012 +csv +cuda +cuda12 +cudnn9 +cvzf +Cyberinfrastructure +Cygwin +d8c +d8f +Dataproc +Dataproc's +DBI +debian +Debrunner +devicelogin +df +dmel +dnf +docker +DOI +DOIs +doisearch +dos2unix +dplyr +Drosophila +drwx +drwxr +drwxrwxr +DS3 +DTN +dtn +DTNs +DUA +DUAs +editfacl +edu +el9 +eleuther +emacs +Emeriti +ENV +env +envs +Esc +ExitCode +f +FACL +faidx +FAIpQLSdehngqL1xso +fb +FBgn0000042 +fc +FDT +FFTW +fftw +file1 +file2 +fileToSearch +FileZilla +FlyBase +FORC +FORTRAN +ForwardAgent +FP16 +fs +G. +ga +Gb +gcc +GCP +GCP's +GCS +gDTNs +Geene +gemini +getenv +getfacl +GFF3 +gff3 +GID +gif +GitBash +glibc +GLIBCXX +globus +gpt +GQA +gra533 +gra752 +gtf +GTF +GTF2 +Guanaco +gz +h. +HDFS +hdfs +hisat2 +HISTCONTROL +HISTSIZE +HiveQL +Horovod +hpc +HSRN +htslib +huggingface +huggingface’s +I2 +I2CC +igv +IGV +IIQ +ILSVRC +ImageNet’s +ims +ins +Internet2 +ipa +ipykernel +IRB +iso +isoforms +ItemType +j. +JBrowse +jeff +JobID +JOBID +JobName +json +JSONLines +JsonSerDe +K. +Kaggle +kb +Keras +kubernetes +LABNAME +Laiba +langchain +LESSOPEN +libmambapy +LIBPATHS +linux64 +llamaindex +llm +llvm +LMOD +lmod +lofreq +LogLevel +lp +lr +LR +lua +MacOS +macs2 +macs3 +mafft +Mathworks +MathWorks +mct +md +MD +mdw303 +mdw303test +Mehnaz +melanogaster +mem +meminfo +Miniconda +Miniforge +miniforge3 +Miniforge3 +MJSynth +mkdir +MKL +mmfs +MoabXterm +MobaKeyGen +MobaXterm +MobaXTerm +ModelCatalog +ModelCtalog +mozallowfullscreen +mozilla +MPI +mpi4py +mpiexec +MPS +Multi-Factor +multi-processor +multi-step +multi-user +MultiWorkerMirroredStrategy +MXNet +MYchdSIR4EFcYeOthelfQ +n. +n0dx20hk701 +n2 +n3 +NA +nano +Nano +ncpus +NDR400 +netcdf +NETID +netid +NetID1 +NetID2 +NetID3 +netrc +Nextflow +nfs4 +NFS4 +NFSv4 +NIST +NODELIST +NodeManagers +non-active +non-commercial +Non-Linear +nproc +ntasks +NUM +numberOfLines +nv +nvidia +o +O. +o4 +OK +OMP +OnDemand +OpenAssistant +openmpi +OpenShift +openVPN +OpenWebUI +os +OSC +out +Doing +Packrat +partition1 +pc +penv +perl +PID +pnas +pnetcdf +Portkey +Post-Processing +PostreSQL +powershell +pre-built +pre-compiled +pre-installation +pre-installed +pre-requisite +prem +printenv +project1 +ProQuest +ps +pwm +py3 +Pythia +python3 +PYTHONNOUSERSITE +PYTHONPATH +QualityOfService +Qualtrics +Quickconnect +Qwen +r6 +ra +raz +Rclone +rclone +re-running +redownloading +refSeqs +Renviron +requestor’s +reshape2 +ResNet +resrtictions +rf +rhel +rmdir +rp +Rprofile +Rscript +rsqlite +rsync +Rsync +RTS +rwx +RWX +rwxrwxr +rwxrwxrwx +RX +S3 +Sabastian +samtools +Sawe +sbin +scancel +scheduler’s +scp +segmentations +SELINUX +semi-structured +SerDe +ServerAliveInterval +setfacl +SFTP +sftp +SGLang +sgtatham +sha256 +SHLVL +sif +Simulink +Slurm +slurm +slurmstepd +Slurm’s +Solaris +sp +SQLiteStudio +squeue +sr +src +SRR307023 +SRR307024 +SRR307025 +SRR307026 +SRR307027 +SRR307028 +SRR307029 +SRR307030 +SSO +Stata's +STEAMROOT +Storable +StrictHostKeyChecking +sub-directories +sub-directory +subdir +subdir2 +sudo +susie +susie's +svc +svg +sw77 +Sxxxx +Sys +sys +Syslabs +sysparm +System32 +szip +T16 +t97br4zzvip +TAH +TBD +tc001 +TCL +tf +tgz +THB3P2vrDKcg +tibble +Tigst +tmp +tok +Top500 +Tos +Trino +tsqueryId +tsv +txt +u +U4Vmh2Rw4idTcwqzLhNX1g +UID +un-loads +un-tar +unix +UNIX +unix2dos +up +Keep +url +urlTemplate +UserKnownHostsFile +usernm +usp +usr +uv +ux +v. +v1 +v5 +VASP +VCF +vcf +venv +vertexai +VertexAI +vglrun +virtualenv +vm +w +wc +webkitallowfullscreen +webpage +Webpage +webpages +Wget +wget +whatToFind +WiFi +win10 +Windowns +WinSCP +wo +WordNet +wq +wsl +WSL2 +WSP +x. +X11 +x11 +x32 +x7 +xa0 +xa020242026 +XDG +xfs +XQuartz +xr +xvf +xvzf +XXXXSgS6Tp +xzf +yml +yourNetID +YV6MOhplKNwxXjASHYnDtM +zcat +zfs +zU +Zvvga2Y8kQZN +zypper From 0ab4847d7e87a890e6cfa253634311d8ad96f259 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 12:58:31 +0800 Subject: [PATCH 15/37] Modify Harper workflow to check selected rules Updated the workflow to check selected Harper rules instead of running the Harper CLI directly. Added Python script to process and report selected lints. --- .github/workflows/harper.yml | 48 +++++++++++++++++++++++++++++++++--- 1 file changed, 44 insertions(+), 4 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index d9d197a457..6499f4382a 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -45,8 +45,48 @@ jobs: print(word) PY - - name: Run Harper + - name: Check Selected Harper Rules run: | - find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ - xargs -0 harper-cli lint \ - --user-dict-path ./.harper-dictionary.txt || true + python - <<'PY' + import json + import sys + from pathlib import Path + + with open("harper-report.json", encoding="utf-8") as f: + reports = json.load(f) + + selected = [] + + for report in reports: + file = report.get("file") + path = Path(file) + lines = path.read_text(encoding="utf-8", errors="ignore").splitlines() if path.exists() else [] + + for lint in report.get("lints", []): + kind = lint.get("kind") + rule = lint.get("rule") + line_no = lint.get("line") + message = lint.get("message") + text = lint.get("matched_text") + + keep = False + + if kind in {"Spelling", "Grammar", "Repetition"}: + keep = True + + if kind == "Capitalization" and rule == "UseTitleCase": + if line_no and 1 <= line_no <= len(lines): + if lines[line_no - 1].lstrip().startswith("#"): + keep = True + + if keep: + selected.append((file, line_no, kind, rule, text, message)) + + if selected: + print("Selected Harper lints found:") + for file, line, kind, rule, text, message in selected: + print(f"{file}:{line}: [{kind}::{rule}] {text!r} - {message}") + sys.exit(1) + + print("No selected Harper lints found.") + PY From 03a415686e524a99f59c5747053e67e58a887b96 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 14:06:44 +0800 Subject: [PATCH 16/37] Update harper.yml --- .github/workflows/harper.yml | 35 +++++++++++++++++++++++++++-------- 1 file changed, 27 insertions(+), 8 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 6499f4382a..4558fb9ad8 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -33,6 +33,7 @@ jobs: reports = json.load(f) words = set() + for report in reports: for lint in report.get("lints", []): if lint.get("kind") == "Spelling": @@ -40,9 +41,12 @@ jobs: if word: words.add(word) - print("Suggested dictionary candidates:") - for word in sorted(words, key=str.lower): - print(word) + with open("dictionary-candidates.txt", "w", encoding="utf-8") as out: + out.write("Suggested dictionary candidates:\n") + for word in sorted(words, key=str.lower): + out.write(word + "\n") + + print(open("dictionary-candidates.txt", encoding="utf-8").read()) PY - name: Check Selected Harper Rules @@ -59,8 +63,8 @@ jobs: for report in reports: file = report.get("file") - path = Path(file) - lines = path.read_text(encoding="utf-8", errors="ignore").splitlines() if path.exists() else [] + path = Path(file) if file else None + lines = path.read_text(encoding="utf-8", errors="ignore").splitlines() if path and path.exists() else [] for lint in report.get("lints", []): kind = lint.get("kind") @@ -80,13 +84,28 @@ jobs: keep = True if keep: - selected.append((file, line_no, kind, rule, text, message)) + selected.append( + f"{file}:{line_no}: [{kind}::{rule}] {text!r} - {message}" + ) + + with open("filtered-harper-report.txt", "w", encoding="utf-8") as out: + for item in selected: + out.write(item + "\n") if selected: print("Selected Harper lints found:") - for file, line, kind, rule, text, message in selected: - print(f"{file}:{line}: [{kind}::{rule}] {text!r} - {message}") + print(open("filtered-harper-report.txt", encoding="utf-8").read()) sys.exit(1) print("No selected Harper lints found.") PY + + - name: Upload Harper Reports + if: always() + uses: actions/upload-artifact@v6 + with: + name: harper-reports + path: | + harper-report.json + filtered-harper-report.txt + dictionary-candidates.txt From b30a67da8acee29b657d3c1f12690e1fe1684e42 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 15:14:07 +0800 Subject: [PATCH 17/37] Update harper.yml --- .github/workflows/harper.yml | 58 ++++++++++++++---------------------- 1 file changed, 23 insertions(+), 35 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 4558fb9ad8..1e51808d1f 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -13,8 +13,25 @@ jobs: - name: Checkout uses: actions/checkout@v6 + - name: Cache Cargo + uses: actions/cache@v4 + with: + path: | + ~/.cargo/bin + ~/.cargo/registry + ~/.cargo/git + target + key: ${{ runner.os }}-cargo-harper-${{ hashFiles('.github/workflows/harper.yml') }} + restore-keys: | + ${{ runner.os }}-cargo-harper- + - name: Install Harper CLI - run: cargo install --git https://github.com/Automattic/harper harper-cli --locked + run: | + if ! command -v harper-cli >/dev/null 2>&1; then + cargo install --git https://github.com/Automattic/harper harper-cli --locked + else + echo "Harper CLI found in cache." + fi - name: Generate Harper JSON Report run: | @@ -41,12 +58,9 @@ jobs: if word: words.add(word) - with open("dictionary-candidates.txt", "w", encoding="utf-8") as out: - out.write("Suggested dictionary candidates:\n") - for word in sorted(words, key=str.lower): - out.write(word + "\n") - - print(open("dictionary-candidates.txt", encoding="utf-8").read()) + print("Suggested dictionary candidates:") + for word in sorted(words, key=str.lower): + print(word) PY - name: Check Selected Harper Rules @@ -54,7 +68,6 @@ jobs: python - <<'PY' import json import sys - from pathlib import Path with open("harper-report.json", encoding="utf-8") as f: reports = json.load(f) @@ -63,8 +76,6 @@ jobs: for report in reports: file = report.get("file") - path = Path(file) if file else None - lines = path.read_text(encoding="utf-8", errors="ignore").splitlines() if path and path.exists() else [] for lint in report.get("lints", []): kind = lint.get("kind") @@ -73,39 +84,16 @@ jobs: message = lint.get("message") text = lint.get("matched_text") - keep = False - if kind in {"Spelling", "Grammar", "Repetition"}: - keep = True - - if kind == "Capitalization" and rule == "UseTitleCase": - if line_no and 1 <= line_no <= len(lines): - if lines[line_no - 1].lstrip().startswith("#"): - keep = True - - if keep: selected.append( f"{file}:{line_no}: [{kind}::{rule}] {text!r} - {message}" ) - with open("filtered-harper-report.txt", "w", encoding="utf-8") as out: - for item in selected: - out.write(item + "\n") - if selected: print("Selected Harper lints found:") - print(open("filtered-harper-report.txt", encoding="utf-8").read()) + for item in selected: + print(item) sys.exit(1) print("No selected Harper lints found.") PY - - - name: Upload Harper Reports - if: always() - uses: actions/upload-artifact@v6 - with: - name: harper-reports - path: | - harper-report.json - filtered-harper-report.txt - dictionary-candidates.txt From 25a3a241ae96a31e1911b27befc4fe727539fdfe Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 15:23:54 +0800 Subject: [PATCH 18/37] Update .harper-dictionary.txt --- .harper-dictionary.txt | 58 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 58 insertions(+) diff --git a/.harper-dictionary.txt b/.harper-dictionary.txt index f4f6fe2f53..fe4788a0b4 100644 --- a/.harper-dictionary.txt +++ b/.harper-dictionary.txt @@ -680,3 +680,61 @@ zfs zU Zvvga2Y8kQZN zypper +anyconnect +autoscaling +bam +behaviour +behaviours +cancelled +CANCELLED +cancelling +Cancelling +catalogue +Catalogue +column1 +column2 +column3 +column4 +column6 +column7 +Customising +facl +forc +gtf +HH +huggingface’s +igv +ImageNet’s +jbrowse +JobID +kevin +kevyin +labelled +lefthand +libPaths +lr +Mathworks +md +miniforge +miniforge3 +MobaXterm +mpi +nano +nextflow +nfs4 +OK +ondemand +out +Doing +pid +portkey +rsync +rwx +scheduler’s +Slurm’s +un +unix +up +Keep +vertexai +webpage From 5de2f1360fe507b2c763bb16f5e07ffd7d577849 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 15:55:56 +0800 Subject: [PATCH 19/37] Update .harper-dictionary.txt --- .harper-dictionary.txt | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/.harper-dictionary.txt b/.harper-dictionary.txt index fe4788a0b4..2f3d966ba6 100644 --- a/.harper-dictionary.txt +++ b/.harper-dictionary.txt @@ -382,8 +382,6 @@ MacOS macs2 macs3 mafft -Mathworks -MathWorks mct md MD @@ -738,3 +736,11 @@ up Keep vertexai webpage +approver +requestor’s +ImageNet’s +huggingface’s +Slurm’s +MobaXTerm +Assefa +JOBID From d429fe719aa7a88f5489d9bbad2b1af8a5aa7e20 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 16:12:46 +0800 Subject: [PATCH 20/37] Update harper.yml --- .github/workflows/harper.yml | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 1e51808d1f..42913894c3 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -84,10 +84,12 @@ jobs: message = lint.get("message") text = lint.get("matched_text") - if kind in {"Spelling", "Grammar", "Repetition"}: - selected.append( - f"{file}:{line_no}: [{kind}::{rule}] {text!r} - {message}" - ) + if kind == "Spelling" and rule == "SpellCheck": + keep = True + if kind == "Grammar": + keep = True + if kind == "Repetition": + keep = True if selected: print("Selected Harper lints found:") From 258b0b30be2f38692350c083604b7eb20987c87c Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Fri, 3 Jul 2026 22:10:08 +0800 Subject: [PATCH 21/37] Refactor Harper CI workflow and linting logic --- .github/workflows/harper.yml | 39 ++++++++++++++++++++++++------------ 1 file changed, 26 insertions(+), 13 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 42913894c3..3c6644f74b 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -20,10 +20,9 @@ jobs: ~/.cargo/bin ~/.cargo/registry ~/.cargo/git - target - key: ${{ runner.os }}-cargo-harper-${{ hashFiles('.github/workflows/harper.yml') }} + key: ${{ runner.os }}-cargo-harper restore-keys: | - ${{ runner.os }}-cargo-harper- + ${{ runner.os }}-cargo-harper - name: Install Harper CLI run: | @@ -53,10 +52,12 @@ jobs: for report in reports: for lint in report.get("lints", []): - if lint.get("kind") == "Spelling": - word = lint.get("matched_text") - if word: - words.add(word) + kind = lint.get("kind") + rule = lint.get("rule") + word = lint.get("matched_text") + + if kind == "Spelling" and rule == "SpellCheck" and word: + words.add(word) print("Suggested dictionary candidates:") for word in sorted(words, key=str.lower): @@ -73,23 +74,35 @@ jobs: reports = json.load(f) selected = [] + total_lints = 0 + + IGNORED_RULES = { + ("Spelling", "OkToOkay"), + ("Spelling", "DisjointPrefixes"), + } for report in reports: file = report.get("file") for lint in report.get("lints", []): + total_lints += 1 + kind = lint.get("kind") rule = lint.get("rule") line_no = lint.get("line") message = lint.get("message") text = lint.get("matched_text") - if kind == "Spelling" and rule == "SpellCheck": - keep = True - if kind == "Grammar": - keep = True - if kind == "Repetition": - keep = True + if (kind, rule) in IGNORED_RULES: + continue + + if kind in {"Spelling", "Grammar", "Repetition"}: + selected.append( + f"{file}:{line_no}: [{kind}::{rule}] {text!r} - {message}" + ) + + print(f"Total Harper lint count: {total_lints}") + print(f"Selected lint count: {len(selected)}") if selected: print("Selected Harper lints found:") From f9d0e3c06464774065c733f450b386323b760e9c Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Mon, 6 Jul 2026 23:08:23 +0800 Subject: [PATCH 22/37] Enhance Harper rule check with ignore conditions Added a function to ignore certain spelling texts based on specific criteria. --- .github/workflows/harper.yml | 32 ++++++++++++++++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 3c6644f74b..8e176743f4 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -68,6 +68,7 @@ jobs: run: | python - <<'PY' import json + import re import sys with open("harper-report.json", encoding="utf-8") as f: @@ -81,6 +82,34 @@ jobs: ("Spelling", "DisjointPrefixes"), } + def should_ignore_spelling_text(text): + if not text: + return False + + stripped = text.strip() + + # URLs and URL fragments + if stripped.startswith(("http://", "https://", "www.")): + return True + + # File paths / command paths + if "/" in stripped: + return True + + # Long hashes, IDs, tokens + if len(stripped) >= 16 and any(c.isdigit() for c in stripped): + return True + + # Hex-like strings, e.g. d5442a5dc4baadd48b32 + if len(stripped) >= 8 and re.fullmatch(r"[a-fA-F0-9]+", stripped): + return True + + # Mostly random-looking alphanumeric tokens + if len(stripped) >= 12 and re.fullmatch(r"[A-Za-z0-9_-]+", stripped): + return True + + return False + for report in reports: file = report.get("file") @@ -96,6 +125,9 @@ jobs: if (kind, rule) in IGNORED_RULES: continue + if kind == "Spelling" and should_ignore_spelling_text(text): + continue + if kind in {"Spelling", "Grammar", "Repetition"}: selected.append( f"{file}:{line_no}: [{kind}::{rule}] {text!r} - {message}" From b967a45069560a0d99ce28c2cd0a51476ada25a4 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Mon, 6 Jul 2026 23:13:23 +0800 Subject: [PATCH 23/37] Update .harper-dictionary.txt --- .harper-dictionary.txt | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/.harper-dictionary.txt b/.harper-dictionary.txt index 2f3d966ba6..e620e68e79 100644 --- a/.harper-dictionary.txt +++ b/.harper-dictionary.txt @@ -687,8 +687,6 @@ cancelled CANCELLED cancelling Cancelling -catalogue -Catalogue column1 column2 column3 @@ -744,3 +742,19 @@ Slurm’s MobaXTerm Assefa JOBID +ImageNet +Acknowledgements +afterwards +Customizing +nnodes +CANCELLED +Cancelling +cancelling +cancelled +behaviour +Backfilling +favourite +MobaXterm +flavours +labelled +behaviours From 2774f7a17568399e2c1cd490276ff666e58414e0 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Mon, 6 Jul 2026 23:21:38 +0800 Subject: [PATCH 24/37] Add 'analyses' to the dictionary --- .harper-dictionary.txt | 1 + 1 file changed, 1 insertion(+) diff --git a/.harper-dictionary.txt b/.harper-dictionary.txt index e620e68e79..152e4462d3 100644 --- a/.harper-dictionary.txt +++ b/.harper-dictionary.txt @@ -28,6 +28,7 @@ gDTN conda QuickLinks QuickLink +analyses req cvf rclone From c59ce3445c5eb740f6467f2cba5faaf0e8620bb9 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 18:12:33 +0800 Subject: [PATCH 25/37] turn to capspell for spelling detection --- .harper-dictionary.txt => project-words.txt | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename .harper-dictionary.txt => project-words.txt (100%) diff --git a/.harper-dictionary.txt b/project-words.txt similarity index 100% rename from .harper-dictionary.txt rename to project-words.txt From 65fe2114495e25b8e96e1e876e691e2596e16eac Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 18:13:17 +0800 Subject: [PATCH 26/37] Refactor GitHub Actions workflow for documentation quality --- .github/workflows/harper.yml | 139 +++++++++++++++++++++++++---------- 1 file changed, 102 insertions(+), 37 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 8e176743f4..7ce7afcd89 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -1,65 +1,130 @@ -name: Harper +name: Docs Quality on: pull_request: workflow_dispatch: + inputs: + allow_failure: + type: boolean + default: false jobs: - harper: - name: Grammar and Style Check + spelling: runs-on: ubuntu-latest - steps: - - name: Checkout - uses: actions/checkout@v6 + - uses: actions/checkout@v6 + - uses: actions/setup-node@v4 + with: + node-version: "20" + - uses: actions/cache@v4 + with: + path: ~/.npm + key: ${{ runner.os }}-npm-cspell-${{ hashFiles('package-lock.json') }} + restore-keys: ${{ runner.os }}-npm-cspell- + - run: npm ci + - run: | + npx cspell lint \ + "docs/**/*.{md,mdx}" \ + "src/pages/**/*.{md,mdx}" \ + --config cspell.json \ + --no-progress \ + --reporter json \ + > cspell-report.json || true + - uses: actions/upload-artifact@v4 + with: + name: cspell-report + path: cspell-report.json - - name: Cache Cargo - uses: actions/cache@v4 + grammar: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + - uses: actions/cache@v4 with: path: | ~/.cargo/bin ~/.cargo/registry ~/.cargo/git key: ${{ runner.os }}-cargo-harper - restore-keys: | - ${{ runner.os }}-cargo-harper - - - name: Install Harper CLI - run: | - if ! command -v harper-cli >/dev/null 2>&1; then - cargo install --git https://github.com/Automattic/harper harper-cli --locked - else - echo "Harper CLI found in cache." - fi - - - name: Generate Harper JSON Report - run: | + restore-keys: ${{ runner.os }}-cargo-harper + - run: | + command -v harper-cli >/dev/null 2>&1 || \ + cargo install --git https://github.com/Automattic/harper \ + --rev \ + harper-cli --locked + - run: | find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ - xargs -0 harper-cli lint \ - --user-dict-path ./.harper-dictionary.txt \ - --format json \ - > harper-report.json || true + xargs -0 harper-cli lint --format json > harper-report.json || true + - uses: actions/upload-artifact@v4 + with: + name: harper-report + path: harper-report.json - - name: Extract Spelling Candidates + heading-case: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + - id: check + continue-on-error: true + run: python check_heading_case.py docs src/pages + - run: echo "${{ steps.check.outcome }}" > heading-case-result.txt + - uses: actions/upload-artifact@v4 + with: + name: heading-case-result + path: heading-case-result.txt + + aggregate: + needs: [spelling, grammar, heading-case] + runs-on: ubuntu-latest + steps: + - uses: actions/download-artifact@v4 + - env: + ALLOW_FAILURE: ${{ github.event_name == 'workflow_dispatch' && inputs.allow_failure }} run: | python - <<'PY' import json + import os + import sys - with open("harper-report.json", encoding="utf-8") as f: - reports = json.load(f) + def load_json(path): + try: + with open(path, encoding="utf-8") as f: + return json.load(f) + except FileNotFoundError: + return None - words = set() + issues = [] - for report in reports: + cspell_data = load_json("cspell-report/cspell-report.json") + for issue in (cspell_data or {}).get("issues", []): + issues.append( + f"[Spelling] {issue.get('uri')}:{issue.get('row')}: " + f"{issue.get('text')!r} - {issue.get('message', 'unknown word')}" + ) + + ALLOWED_KINDS = {"Grammar", "Repetition"} + for report in load_json("harper-report/harper-report.json") or []: + file = report.get("file") for lint in report.get("lints", []): kind = lint.get("kind") - rule = lint.get("rule") - word = lint.get("matched_text") - - if kind == "Spelling" and rule == "SpellCheck" and word: - words.add(word) - - print("Suggested dictionary candidates:") + if kind not in ALLOWED_KINDS: + continue + issues.append( + f"[{kind}::{lint.get('rule')}] {file}:{lint.get('line')}: " + f"{lint.get('matched_text')!r} - {lint.get('message')}" + ) + + with open("heading-case-result/heading-case-result.txt", encoding="utf-8") as f: + if f.read().strip() == "failure": + issues.append("[HeadingCase] Title case violations found, see heading-case job log.") + + for item in issues: + print(item) + print(f"\n{len(issues)} issue(s) found.") + + allow_failure = os.environ.get("ALLOW_FAILURE") == "true" + sys.exit(1 if issues and not allow_failure else 0) + PY print("Suggested dictionary candidates:") for word in sorted(words, key=str.lower): print(word) PY From 6db8c1e232dff478aed463f8bc39e8812a9a0855 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 18:21:56 +0800 Subject: [PATCH 27/37] Add script to check and fix Markdown heading case This script checks and fixes the title case of ATX headings in Markdown files, allowing for an optional fix mode that modifies the files directly. --- check_heading_case.py | 157 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 157 insertions(+) create mode 100644 check_heading_case.py diff --git a/check_heading_case.py b/check_heading_case.py new file mode 100644 index 0000000000..e1699cf385 --- /dev/null +++ b/check_heading_case.py @@ -0,0 +1,157 @@ +import argparse +import re +import sys +from pathlib import Path + +SMALL_WORDS = { + "a", "an", "the", + "and", "or", "but", "nor", "so", "yet", + "as", "at", "by", "for", "from", "in", "into", + "of", "on", "onto", "over", "per", "to", "up", "via", "with", +} + +ATX_HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*$") +FENCE_RE = re.compile(r"^\s*(```|~~~)") +FRONTMATTER_DELIM_RE = re.compile(r"^---\s*$") +HEADING_ANCHOR_RE = re.compile(r"\s*\{#[\w-]+\}\s*$") +INLINE_CODE_RE = re.compile(r"`[^`]*`") + + +def split_words_preserving_delims(text): + tokens = [] + for part in re.split(r"(\s+|-)", text): + if part == "": + continue + is_word = bool(re.match(r"^[A-Za-z][A-Za-z0-9'’]*$", part)) + tokens.append((part, is_word)) + return tokens + + +def title_case_word(word, is_first, is_last): + lower = word.lower() + if not is_first and not is_last and lower in SMALL_WORDS: + return lower + if word[0].isupper(): + return word + return word[0].upper() + word[1:] + + +def check_and_fix_heading(text): + tokens = split_words_preserving_delims(text) + word_positions = [i for i, (_, is_word) in enumerate(tokens) if is_word] + if not word_positions: + return False, text + + first_idx = word_positions[0] + last_idx = word_positions[-1] + + new_tokens = list(tokens) + changed = False + for i, (tok, is_word) in enumerate(tokens): + if not is_word: + continue + fixed = title_case_word(tok, i == first_idx, i == last_idx) + if fixed != tok: + changed = True + new_tokens[i] = (fixed, True) + + suggestion = "".join(t for t, _ in new_tokens) + return changed, suggestion + + +def extract_headings(lines): + in_frontmatter = False + in_fence = False + results = [] + + for i, line in enumerate(lines): + if i == 0 and FRONTMATTER_DELIM_RE.match(line): + in_frontmatter = True + continue + if in_frontmatter: + if FRONTMATTER_DELIM_RE.match(line): + in_frontmatter = False + continue + + if FENCE_RE.match(line): + in_fence = not in_fence + continue + if in_fence: + continue + + m = ATX_HEADING_RE.match(line) + if not m: + continue + + hashes, raw_text = m.groups() + text_for_check = HEADING_ANCHOR_RE.sub("", raw_text) + code_spans = INLINE_CODE_RE.findall(text_for_check) + placeholder_text = INLINE_CODE_RE.sub("\x00", text_for_check) + + results.append((i, hashes, raw_text, placeholder_text, code_spans)) + + return results + + +def restore_code_spans(text, code_spans): + for span in code_spans: + text = text.replace("\x00", span, 1) + return text + + +def process_file(path, fix=False): + lines = path.read_text(encoding="utf-8").splitlines() + headings = extract_headings(lines) + issues = [] + + for lineno, hashes, raw_text, placeholder_text, code_spans in headings: + changed, suggestion = check_and_fix_heading(placeholder_text) + if not changed: + continue + suggestion = restore_code_spans(suggestion, code_spans) + issues.append((lineno, hashes, raw_text, suggestion)) + + if fix: + lines[lineno] = f"{hashes} {suggestion}" + + if fix and issues: + path.write_text("\n".join(lines) + "\n", encoding="utf-8") + + return issues + + +def main(): + parser = argparse.ArgumentParser(description="Check Markdown ATX heading title case") + parser.add_argument("paths", nargs="+") + parser.add_argument("--fix", action="store_true") + args = parser.parse_args() + + md_files = [] + for p in args.paths: + root = Path(p) + if not root.exists(): + continue + md_files.extend(root.rglob("*.md")) + md_files.extend(root.rglob("*.mdx")) + + total_issues = 0 + for path in sorted(md_files): + issues = process_file(path, fix=args.fix) + for lineno, hashes, raw_text, suggestion in issues: + total_issues += 1 + print(f"{path}:{lineno + 1}: {hashes} {raw_text}") + print(f" suggestion -> {hashes} {suggestion}") + + if total_issues == 0: + print("All headings pass title case check.") + return 0 + + print(f"\n{total_issues} heading case issue(s) found.") + if args.fix: + print("Files were auto-fixed. Please review the diff.") + return 0 + return 1 + + +if __name__ == "__main__": + sys.exit(main()) From d7fe40d513bee07d9a0c6b2910f07b6ea4659cbf Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 18:27:11 +0800 Subject: [PATCH 28/37] Update harper.yml --- .github/workflows/harper.yml | 192 +++++++++-------------------------- 1 file changed, 48 insertions(+), 144 deletions(-) diff --git a/.github/workflows/harper.yml b/.github/workflows/harper.yml index 7ce7afcd89..1817823589 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/harper.yml @@ -8,6 +8,9 @@ on: type: boolean default: false +env: + ALLOW_FAILURE: ${{ github.event_name == 'workflow_dispatch' && inputs.allow_failure }} + jobs: spelling: runs-on: ubuntu-latest @@ -19,21 +22,37 @@ jobs: - uses: actions/cache@v4 with: path: ~/.npm - key: ${{ runner.os }}-npm-cspell-${{ hashFiles('package-lock.json') }} + key: ${{ runner.os }}-npm-cspell restore-keys: ${{ runner.os }}-npm-cspell- - - run: npm ci - run: | - npx cspell lint \ + npx --yes cspell@9 lint \ "docs/**/*.{md,mdx}" \ "src/pages/**/*.{md,mdx}" \ --config cspell.json \ --no-progress \ --reporter json \ > cspell-report.json || true - - uses: actions/upload-artifact@v4 - with: - name: cspell-report - path: cspell-report.json + - run: | + python - <<'PY' + import json + import os + import sys + + with open("cspell-report.json", encoding="utf-8") as f: + data = json.load(f) + + issues = data.get("issues", []) + for issue in issues: + file = issue.get("uri") + line = issue.get("row") + message = issue.get("message") or f"Unknown word: {issue.get('text')!r}" + print(f"::error file={file},line={line}::{message}") + + print(f"{len(issues)} spelling issue(s) found.") + + allow_failure = os.environ.get("ALLOW_FAILURE") == "true" + sys.exit(1 if issues and not allow_failure else 0) + PY grammar: runs-on: ubuntu-latest @@ -50,162 +69,47 @@ jobs: - run: | command -v harper-cli >/dev/null 2>&1 || \ cargo install --git https://github.com/Automattic/harper \ - --rev \ + --tag v2.1.0 \ harper-cli --locked - run: | find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ xargs -0 harper-cli lint --format json > harper-report.json || true - - uses: actions/upload-artifact@v4 - with: - name: harper-report - path: harper-report.json - - heading-case: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v6 - - id: check - continue-on-error: true - run: python check_heading_case.py docs src/pages - - run: echo "${{ steps.check.outcome }}" > heading-case-result.txt - - uses: actions/upload-artifact@v4 - with: - name: heading-case-result - path: heading-case-result.txt - - aggregate: - needs: [spelling, grammar, heading-case] - runs-on: ubuntu-latest - steps: - - uses: actions/download-artifact@v4 - - env: - ALLOW_FAILURE: ${{ github.event_name == 'workflow_dispatch' && inputs.allow_failure }} - run: | + - run: | python - <<'PY' import json import os import sys - def load_json(path): - try: - with open(path, encoding="utf-8") as f: - return json.load(f) - except FileNotFoundError: - return None + with open("harper-report.json", encoding="utf-8") as f: + reports = json.load(f) + ALLOWED_KINDS = {"Grammar", "Repetition"} issues = [] - cspell_data = load_json("cspell-report/cspell-report.json") - for issue in (cspell_data or {}).get("issues", []): - issues.append( - f"[Spelling] {issue.get('uri')}:{issue.get('row')}: " - f"{issue.get('text')!r} - {issue.get('message', 'unknown word')}" - ) - - ALLOWED_KINDS = {"Grammar", "Repetition"} - for report in load_json("harper-report/harper-report.json") or []: + for report in reports: file = report.get("file") for lint in report.get("lints", []): - kind = lint.get("kind") - if kind not in ALLOWED_KINDS: + if lint.get("kind") not in ALLOWED_KINDS: continue - issues.append( - f"[{kind}::{lint.get('rule')}] {file}:{lint.get('line')}: " - f"{lint.get('matched_text')!r} - {lint.get('message')}" - ) - - with open("heading-case-result/heading-case-result.txt", encoding="utf-8") as f: - if f.read().strip() == "failure": - issues.append("[HeadingCase] Title case violations found, see heading-case job log.") + line = lint.get("line") + message = lint.get("message") + print(f"::error file={file},line={line}::{message}") + issues.append(lint) - for item in issues: - print(item) - print(f"\n{len(issues)} issue(s) found.") + print(f"{len(issues)} grammar/repetition issue(s) found.") allow_failure = os.environ.get("ALLOW_FAILURE") == "true" sys.exit(1 if issues and not allow_failure else 0) - PY print("Suggested dictionary candidates:") - for word in sorted(words, key=str.lower): - print(word) PY - - name: Check Selected Harper Rules - run: | - python - <<'PY' - import json - import re - import sys - - with open("harper-report.json", encoding="utf-8") as f: - reports = json.load(f) - - selected = [] - total_lints = 0 - - IGNORED_RULES = { - ("Spelling", "OkToOkay"), - ("Spelling", "DisjointPrefixes"), + heading-case: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + - run: | + python check_heading_case.py docs src/pages || { + if [ "$ALLOW_FAILURE" = "true" ]; then + exit 0 + fi + exit 1 } - - def should_ignore_spelling_text(text): - if not text: - return False - - stripped = text.strip() - - # URLs and URL fragments - if stripped.startswith(("http://", "https://", "www.")): - return True - - # File paths / command paths - if "/" in stripped: - return True - - # Long hashes, IDs, tokens - if len(stripped) >= 16 and any(c.isdigit() for c in stripped): - return True - - # Hex-like strings, e.g. d5442a5dc4baadd48b32 - if len(stripped) >= 8 and re.fullmatch(r"[a-fA-F0-9]+", stripped): - return True - - # Mostly random-looking alphanumeric tokens - if len(stripped) >= 12 and re.fullmatch(r"[A-Za-z0-9_-]+", stripped): - return True - - return False - - for report in reports: - file = report.get("file") - - for lint in report.get("lints", []): - total_lints += 1 - - kind = lint.get("kind") - rule = lint.get("rule") - line_no = lint.get("line") - message = lint.get("message") - text = lint.get("matched_text") - - if (kind, rule) in IGNORED_RULES: - continue - - if kind == "Spelling" and should_ignore_spelling_text(text): - continue - - if kind in {"Spelling", "Grammar", "Repetition"}: - selected.append( - f"{file}:{line_no}: [{kind}::{rule}] {text!r} - {message}" - ) - - print(f"Total Harper lint count: {total_lints}") - print(f"Selected lint count: {len(selected)}") - - if selected: - print("Selected Harper lints found:") - for item in selected: - print(item) - sys.exit(1) - - print("No selected Harper lints found.") - PY From 978660e29dca9d1ef32191389984a7712a821c70 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 18:32:26 +0800 Subject: [PATCH 29/37] Create cspell.json --- cspell.json | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) create mode 100644 cspell.json diff --git a/cspell.json b/cspell.json new file mode 100644 index 0000000000..8abdd2eedc --- /dev/null +++ b/cspell.json @@ -0,0 +1,20 @@ +{ + "version": "0.2", + "language": "en", + "dictionaries": ["bash", "npm", "node", "softwareTerms"], + "dictionaryDefinitions": [ + { + "name": "project-words", + "path": "./project-words.txt", + "addWords": true + } + ], + "ignoreRegExpList": [ + "https?:\\/\\/\\S+", + "[a-fA-F0-9]{8,}", + "\\b[A-Za-z0-9_-]{16,}\\b" + ], + "ignorePaths": [ + "**/node_modules/**" + ] +} From f1b07235db9e00704881c37607dfd9871f492231 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 18:41:30 +0800 Subject: [PATCH 30/37] Update and rename harper.yml to docs-quality.yml --- .../{harper.yml => docs-quality.yml} | 29 +++++++++++++++---- 1 file changed, 24 insertions(+), 5 deletions(-) rename .github/workflows/{harper.yml => docs-quality.yml} (78%) diff --git a/.github/workflows/harper.yml b/.github/workflows/docs-quality.yml similarity index 78% rename from .github/workflows/harper.yml rename to .github/workflows/docs-quality.yml index 1817823589..3dc06ce596 100644 --- a/.github/workflows/harper.yml +++ b/.github/workflows/docs-quality.yml @@ -38,8 +38,22 @@ jobs: import os import sys - with open("cspell-report.json", encoding="utf-8") as f: - data = json.load(f) + try: + with open("cspell-report.json", encoding="utf-8") as f: + content = f.read().strip() + except FileNotFoundError: + content = "" + + if not content: + print("::error::cspell-report.json is empty. Check the previous step's log for the actual cspell error (e.g. missing cspell.json or project-words.txt).") + sys.exit(1) + + try: + data = json.loads(content) + except json.JSONDecodeError: + print("::error::cspell-report.json is not valid JSON. Raw output was:") + print(content[:2000]) + sys.exit(1) issues = data.get("issues", []) for issue in issues: @@ -54,12 +68,17 @@ jobs: sys.exit(1 if issues and not allow_failure else 0) PY - grammar: + heading-case: runs-on: ubuntu-latest steps: - uses: actions/checkout@v6 - - uses: actions/cache@v4 - with: + - run: | + python check_heading_case.py docs src/pages || { + if [ "$ALLOW_FAILURE" = "true" ]; then + exit 0 + fi + exit 1 + } with: path: | ~/.cargo/bin ~/.cargo/registry From eec944de390c47dd39c931879a03ac3f010d3db4 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 19:56:07 +0800 Subject: [PATCH 31/37] Update docs-quality.yml --- .github/workflows/docs-quality.yml | 53 ------------------------------ 1 file changed, 53 deletions(-) diff --git a/.github/workflows/docs-quality.yml b/.github/workflows/docs-quality.yml index 3dc06ce596..07c914ff16 100644 --- a/.github/workflows/docs-quality.yml +++ b/.github/workflows/docs-quality.yml @@ -68,59 +68,6 @@ jobs: sys.exit(1 if issues and not allow_failure else 0) PY - heading-case: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v6 - - run: | - python check_heading_case.py docs src/pages || { - if [ "$ALLOW_FAILURE" = "true" ]; then - exit 0 - fi - exit 1 - } with: - path: | - ~/.cargo/bin - ~/.cargo/registry - ~/.cargo/git - key: ${{ runner.os }}-cargo-harper - restore-keys: ${{ runner.os }}-cargo-harper - - run: | - command -v harper-cli >/dev/null 2>&1 || \ - cargo install --git https://github.com/Automattic/harper \ - --tag v2.1.0 \ - harper-cli --locked - - run: | - find docs src/pages -type f \( -name "*.md" -o -name "*.mdx" \) -print0 | \ - xargs -0 harper-cli lint --format json > harper-report.json || true - - run: | - python - <<'PY' - import json - import os - import sys - - with open("harper-report.json", encoding="utf-8") as f: - reports = json.load(f) - - ALLOWED_KINDS = {"Grammar", "Repetition"} - issues = [] - - for report in reports: - file = report.get("file") - for lint in report.get("lints", []): - if lint.get("kind") not in ALLOWED_KINDS: - continue - line = lint.get("line") - message = lint.get("message") - print(f"::error file={file},line={line}::{message}") - issues.append(lint) - - print(f"{len(issues)} grammar/repetition issue(s) found.") - - allow_failure = os.environ.get("ALLOW_FAILURE") == "true" - sys.exit(1 if issues and not allow_failure else 0) - PY - heading-case: runs-on: ubuntu-latest steps: From 5c7e58d4d12ab068c1fa673fbbae1a535778f650 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 22:33:39 +0800 Subject: [PATCH 32/37] Update check_heading_case.py --- check_heading_case.py | 22 ++++++++++++++++------ 1 file changed, 16 insertions(+), 6 deletions(-) diff --git a/check_heading_case.py b/check_heading_case.py index e1699cf385..daa734099f 100644 --- a/check_heading_case.py +++ b/check_heading_case.py @@ -1,8 +1,17 @@ import argparse +import os import re import sys from pathlib import Path +IN_GITHUB_ACTIONS = os.environ.get("GITHUB_ACTIONS") == "true" + + +def emit(path, line, message): + print(f"{path}:{line}: {message}") + if IN_GITHUB_ACTIONS: + print(f"::error file={path},line={line}::{message}") + SMALL_WORDS = { "a", "an", "the", "and", "or", "but", "nor", "so", "yet", @@ -17,12 +26,13 @@ INLINE_CODE_RE = re.compile(r"`[^`]*`") +TOKEN_RE = re.compile(r"[A-Za-z0-9'’]+|[^A-Za-z0-9'’]+") + + def split_words_preserving_delims(text): tokens = [] - for part in re.split(r"(\s+|-)", text): - if part == "": - continue - is_word = bool(re.match(r"^[A-Za-z][A-Za-z0-9'’]*$", part)) + for part in TOKEN_RE.findall(text): + is_word = part[0].isalpha() tokens.append((part, is_word)) return tokens @@ -139,8 +149,8 @@ def main(): issues = process_file(path, fix=args.fix) for lineno, hashes, raw_text, suggestion in issues: total_issues += 1 - print(f"{path}:{lineno + 1}: {hashes} {raw_text}") - print(f" suggestion -> {hashes} {suggestion}") + message = f"Heading case: '{hashes} {raw_text}' should be '{hashes} {suggestion}'" + emit(str(path), lineno + 1, message) if total_issues == 0: print("All headings pass title case check.") From a065e5ff9353ae44a03a806151b52c81a130361c Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 22:49:45 +0800 Subject: [PATCH 33/37] Rename workflow and update steps for docs quality --- .github/workflows/code-quality.yml | 102 +++++++++++++++++++---------- 1 file changed, 67 insertions(+), 35 deletions(-) diff --git a/.github/workflows/code-quality.yml b/.github/workflows/code-quality.yml index bb42866291..46b6f83cbf 100644 --- a/.github/workflows/code-quality.yml +++ b/.github/workflows/code-quality.yml @@ -1,49 +1,81 @@ -name: Code Quality - -permissions: - contents: read +name: Docs Quality on: - push: - branches: [main] pull_request: - branches: [main] + workflow_dispatch: + inputs: + allow_failure: + type: boolean + default: false + +env: + ALLOW_FAILURE: ${{ github.event_name == 'workflow_dispatch' && inputs.allow_failure }} jobs: - code-quality: - name: Code Quality + spelling: runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v7 - - - name: Setup Node.js - uses: actions/setup-node@v6 + - uses: actions/checkout@v6 + - uses: actions/setup-node@v4 with: - node-version: latest - - - name: Install pnpm - uses: pnpm/action-setup@v6 + node-version: "20" + - uses: actions/cache@v4 with: - version: latest + path: ~/.npm + key: ${{ runner.os }}-npm-cspell + restore-keys: ${{ runner.os }}-npm-cspell- + - run: | + npx --yes -p cspell@9 -p @cspell/cspell-json-reporter cspell lint \ + "docs/**/*.{md,mdx}" \ + "src/pages/**/*.{md,mdx}" \ + --config cspell.json \ + --no-progress \ + --reporter "@cspell/cspell-json-reporter" \ + > cspell-report.json || true + - run: | + python - <<'PY' + import json + import os + import sys - - name: Setup pnpm config - run: | - pnpm config set store-dir ~/.pnpm-store + try: + with open("cspell-report.json", encoding="utf-8") as f: + content = f.read().strip() + except FileNotFoundError: + content = "" - - uses: actions/cache@v6 - name: Setup pnpm cache - with: - path: ~/.pnpm-store - key: ${{ runner.os }}-pnpm-store-${{ hashFiles('**/pnpm-lock.yaml') }} - restore-keys: | - ${{ runner.os }}-pnpm-store- + if not content: + print("::error::cspell-report.json is empty. Check the previous step's log for the actual cspell error (e.g. missing cspell.json or project-words.txt).") + sys.exit(1) + + try: + data = json.loads(content) + except json.JSONDecodeError: + print("::error::cspell-report.json is not valid JSON. Raw output was:") + print(content[:2000]) + sys.exit(1) + + issues = data.get("issues", []) + for issue in issues: + file = issue.get("uri") + line = issue.get("row") + message = issue.get("message") or f"Unknown word: {issue.get('text')!r}" + print(f"::error file={file},line={line}::{message}") - - name: Install dependencies - run: pnpm install + print(f"{len(issues)} spelling issue(s) found.") - - name: Check Prettier formatting - run: pnpm format --check + allow_failure = os.environ.get("ALLOW_FAILURE") == "true" + sys.exit(1 if issues and not allow_failure else 0) + PY - - name: Run ESLint and Stylelint - run: pnpm lint + heading-case: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + - run: | + python check_heading_case.py docs src/pages || { + if [ "$ALLOW_FAILURE" = "true" ]; then + exit 0 + fi + exit 1 + } From 899201a5d62d0b4feec36f0987060fd5bf370bbe Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 23:07:01 +0800 Subject: [PATCH 34/37] Update cspell.json From 676b28f1fae65c5b8ddf7ade97fa5f7b41a9f7ce Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 23:08:44 +0800 Subject: [PATCH 35/37] Update cspell.json --- cspell.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cspell.json b/cspell.json index 8abdd2eedc..2da3ae3c1f 100644 --- a/cspell.json +++ b/cspell.json @@ -1,7 +1,7 @@ { "version": "0.2", "language": "en", - "dictionaries": ["bash", "npm", "node", "softwareTerms"], + "dictionaries": ["project-words", "bash", "npm", "node", "softwareTerms"], "dictionaryDefinitions": [ { "name": "project-words", From b9610c7081d92bee1def26ef93a86514ffa38a43 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Tue, 7 Jul 2026 23:32:06 +0800 Subject: [PATCH 36/37] Update docs-quality.yml --- .github/workflows/docs-quality.yml | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/.github/workflows/docs-quality.yml b/.github/workflows/docs-quality.yml index 07c914ff16..9951e4e0ff 100644 --- a/.github/workflows/docs-quality.yml +++ b/.github/workflows/docs-quality.yml @@ -25,12 +25,12 @@ jobs: key: ${{ runner.os }}-npm-cspell restore-keys: ${{ runner.os }}-npm-cspell- - run: | - npx --yes cspell@9 lint \ + npx --yes -p cspell@9 -p @cspell/cspell-json-reporter cspell lint \ "docs/**/*.{md,mdx}" \ "src/pages/**/*.{md,mdx}" \ --config cspell.json \ --no-progress \ - --reporter json \ + --reporter "@cspell/cspell-json-reporter" \ > cspell-report.json || true - run: | python - <<'PY' @@ -57,9 +57,13 @@ jobs: issues = data.get("issues", []) for issue in issues: - file = issue.get("uri") + uri = issue.get("uri", "") + file = uri.replace("file://", "") + if "/rts-docs/" in file: + file = file.split("/rts-docs/", 1)[-1] line = issue.get("row") message = issue.get("message") or f"Unknown word: {issue.get('text')!r}" + print(f"{file}:{line}: {message}") print(f"::error file={file},line={line}::{message}") print(f"{len(issues)} spelling issue(s) found.") From f6e4a1cb0bcb77f755b57ee64753917791074611 Mon Sep 17 00:00:00 2001 From: Amanda-dong <159391549+Amanda-dong@users.noreply.github.com> Date: Wed, 8 Jul 2026 22:56:10 +0800 Subject: [PATCH 37/37] Delete check_heading_case.py --- check_heading_case.py | 167 ------------------------------------------ 1 file changed, 167 deletions(-) delete mode 100644 check_heading_case.py diff --git a/check_heading_case.py b/check_heading_case.py deleted file mode 100644 index daa734099f..0000000000 --- a/check_heading_case.py +++ /dev/null @@ -1,167 +0,0 @@ -import argparse -import os -import re -import sys -from pathlib import Path - -IN_GITHUB_ACTIONS = os.environ.get("GITHUB_ACTIONS") == "true" - - -def emit(path, line, message): - print(f"{path}:{line}: {message}") - if IN_GITHUB_ACTIONS: - print(f"::error file={path},line={line}::{message}") - -SMALL_WORDS = { - "a", "an", "the", - "and", "or", "but", "nor", "so", "yet", - "as", "at", "by", "for", "from", "in", "into", - "of", "on", "onto", "over", "per", "to", "up", "via", "with", -} - -ATX_HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*$") -FENCE_RE = re.compile(r"^\s*(```|~~~)") -FRONTMATTER_DELIM_RE = re.compile(r"^---\s*$") -HEADING_ANCHOR_RE = re.compile(r"\s*\{#[\w-]+\}\s*$") -INLINE_CODE_RE = re.compile(r"`[^`]*`") - - -TOKEN_RE = re.compile(r"[A-Za-z0-9'’]+|[^A-Za-z0-9'’]+") - - -def split_words_preserving_delims(text): - tokens = [] - for part in TOKEN_RE.findall(text): - is_word = part[0].isalpha() - tokens.append((part, is_word)) - return tokens - - -def title_case_word(word, is_first, is_last): - lower = word.lower() - if not is_first and not is_last and lower in SMALL_WORDS: - return lower - if word[0].isupper(): - return word - return word[0].upper() + word[1:] - - -def check_and_fix_heading(text): - tokens = split_words_preserving_delims(text) - word_positions = [i for i, (_, is_word) in enumerate(tokens) if is_word] - if not word_positions: - return False, text - - first_idx = word_positions[0] - last_idx = word_positions[-1] - - new_tokens = list(tokens) - changed = False - for i, (tok, is_word) in enumerate(tokens): - if not is_word: - continue - fixed = title_case_word(tok, i == first_idx, i == last_idx) - if fixed != tok: - changed = True - new_tokens[i] = (fixed, True) - - suggestion = "".join(t for t, _ in new_tokens) - return changed, suggestion - - -def extract_headings(lines): - in_frontmatter = False - in_fence = False - results = [] - - for i, line in enumerate(lines): - if i == 0 and FRONTMATTER_DELIM_RE.match(line): - in_frontmatter = True - continue - if in_frontmatter: - if FRONTMATTER_DELIM_RE.match(line): - in_frontmatter = False - continue - - if FENCE_RE.match(line): - in_fence = not in_fence - continue - if in_fence: - continue - - m = ATX_HEADING_RE.match(line) - if not m: - continue - - hashes, raw_text = m.groups() - text_for_check = HEADING_ANCHOR_RE.sub("", raw_text) - code_spans = INLINE_CODE_RE.findall(text_for_check) - placeholder_text = INLINE_CODE_RE.sub("\x00", text_for_check) - - results.append((i, hashes, raw_text, placeholder_text, code_spans)) - - return results - - -def restore_code_spans(text, code_spans): - for span in code_spans: - text = text.replace("\x00", span, 1) - return text - - -def process_file(path, fix=False): - lines = path.read_text(encoding="utf-8").splitlines() - headings = extract_headings(lines) - issues = [] - - for lineno, hashes, raw_text, placeholder_text, code_spans in headings: - changed, suggestion = check_and_fix_heading(placeholder_text) - if not changed: - continue - suggestion = restore_code_spans(suggestion, code_spans) - issues.append((lineno, hashes, raw_text, suggestion)) - - if fix: - lines[lineno] = f"{hashes} {suggestion}" - - if fix and issues: - path.write_text("\n".join(lines) + "\n", encoding="utf-8") - - return issues - - -def main(): - parser = argparse.ArgumentParser(description="Check Markdown ATX heading title case") - parser.add_argument("paths", nargs="+") - parser.add_argument("--fix", action="store_true") - args = parser.parse_args() - - md_files = [] - for p in args.paths: - root = Path(p) - if not root.exists(): - continue - md_files.extend(root.rglob("*.md")) - md_files.extend(root.rglob("*.mdx")) - - total_issues = 0 - for path in sorted(md_files): - issues = process_file(path, fix=args.fix) - for lineno, hashes, raw_text, suggestion in issues: - total_issues += 1 - message = f"Heading case: '{hashes} {raw_text}' should be '{hashes} {suggestion}'" - emit(str(path), lineno + 1, message) - - if total_issues == 0: - print("All headings pass title case check.") - return 0 - - print(f"\n{total_issues} heading case issue(s) found.") - if args.fix: - print("Files were auto-fixed. Please review the diff.") - return 0 - return 1 - - -if __name__ == "__main__": - sys.exit(main())