From 6c0204d0606f67cf5c9001ac7b93ca41d6e6d1fd Mon Sep 17 00:00:00 2001 From: bruno-ariano Date: Wed, 26 Aug 2026 12:38:15 +0200 Subject: [PATCH 1/4] Adding unit tests --- .coverage | Bin 0 -> 53248 bytes .github/workflows/test.yaml | 41 ++++++++++++++++++ Makefile | 6 +++ run_ingestion_pkgh.sh | 16 ------- tdbsumstat/utils/harmonize_ingest.py | 1 - .../data}/dummy_out_ENSG0000010000.tsv.gz | Bin .../data}/dummy_out_ENSG0000010001.tsv.gz | Bin .../data}/example_data_table.csv | 0 .../data}/locusbreaker_test_table_sc.csv | 0 .../data}/mapping_file_test.csv | 0 {example_data => tests/data}/region_list.csv | 0 .../data}/region_list_sc.csv | 0 {example_data => tests/data}/snp_list_sc.csv | 0 {example_data => tests/data}/trait_list.csv | 0 {example_data => tests/data}/trait_list_r.csv | 0 tests/unit/core/conftest.py | 39 +++++++++++++++++ tests/unit/core/test_ingestion.py | 40 +++++++++++++++++ 17 files changed, 126 insertions(+), 17 deletions(-) create mode 100644 .coverage create mode 100644 .github/workflows/test.yaml delete mode 100644 run_ingestion_pkgh.sh rename {example_data => tests/data}/dummy_out_ENSG0000010000.tsv.gz (100%) rename {example_data => tests/data}/dummy_out_ENSG0000010001.tsv.gz (100%) rename {example_data => tests/data}/example_data_table.csv (100%) rename {example_data => tests/data}/locusbreaker_test_table_sc.csv (100%) rename {example_data => tests/data}/mapping_file_test.csv (100%) rename {example_data => tests/data}/region_list.csv (100%) rename {example_data => tests/data}/region_list_sc.csv (100%) rename {example_data => tests/data}/snp_list_sc.csv (100%) rename {example_data => tests/data}/trait_list.csv (100%) rename {example_data => tests/data}/trait_list_r.csv (100%) create mode 100644 tests/unit/core/conftest.py create mode 100644 tests/unit/core/test_ingestion.py diff --git a/.coverage b/.coverage new file mode 100644 index 0000000000000000000000000000000000000000..7b13a5879d51c52da77d41e2c35d1488b861955f GIT binary patch literal 53248 zcmeI4%WoV-9>=>q?HSLriGnp)R+PGA9m|iw0)&7A}4;jNPB<_7ky_qD56>`;m=&v(pFJf%FQsH*ak5Q?Hy>C>c7zEx;P%Ma+cI5DD9FNN*) zY|9L7cUy!zGS3|avd0r0oujFaGp;A)dC#zWR4SC)Br))9$MpBaJ92L|$&l3+%H1$M zK_b%Lb~%SQofTgezTA?&bS)XAR-Cqb%eCdZ@75~qT_bdcg%2s2WzPz4k#Vk zNa`Ro7UbEsBG=-s60i@JcU$`Awp`Dyro+Oc4kQYljcU1l;pC*y*$)9@V}G|6FO)0o z#YOeKY7$0SYh}MGVm_WZmWr#*tlCH~BgIl*U!UEqO<;HR5EFvA^j&~`pehA+*Q+b_&dGMr5? zHaIQceQPE$`QAc8-=RUDKcz8~`OMJd%W`k+w%MoAWMJApby*sN(-}Z;#-2%;&8`=P zN&NC*5}rKw4DC$?hQ~R3%C($ir|$i#7X&s85i7 zmADj8opvA+8xGD&FAg<=J=WO1y_DFLPcA&nZ&Ihr`xosvj_LNY`HH9ERF%el?wnjt zI*3p~u|%CRpf37=W-gjtIE_%|#VOj$ZZfOg4Qwn*m?7L2FnHTt~{)0k4*gyaTKmY_l z00ck)1V8`;KmY_l00gFwKuMiZE9v_GzM|isJ_q3u2!H?xfB*=900@8p2!H?xfB*>m zIte_cmKGPX{{eV*edEdXr)va4)xT5d2O9{000@8p2!H?xfB*=900@8p2!Oye5vY|G ztJz-x`2K&Z{%1x1n;z=NTOYN$t!DG@O{e)pdbxUT$ehJm0w<(8JH2t{=JHy6M}d>vaaccT-wn(AlJin$MqI3nC{7&9D>py7{iv zw>xro;Q1jv3jNzADJ>LGG7Q_bL&G4g%MH?6ETBbiGm!zmaDl|G)Jg2b5yhg=?gyPM zd)K%bBx;*AQd=vewqyE^=i0xg*T5uZl8x7@q`Fc_HGOW}$RFZo$L9(=KGG(yR!Hqs zArt$a6$M>-zWp8Pv(x3mPDcZJh+mF6eSMZhJ3pJ5#-KmwldW%PB>PMu*{p$jk>l(c z{O$pI5TBHd5-Fc6q`WO%dZ7q$NFMt)EZOfHSs$ee&dyK;)gxBmm^SyJiz;axJEF!& z5b^#0R{aA-|ET^={e#A7J!-nGkDCjPNPoQXdGnK&)oT4R107UB00ck)1V8`;KmY_l z00b1}k`~XaNBm+xw*GHj*5ZXCN+ZjJ<`peoETT2K{%`zFi%%R??7;fJeqD>#im4s4 z{;zFn@k%k(gX{llamPp6WbUdKpDJeJ1MB}vai z{?~43@wsBk4`2V6&S`P=s1;=E|Cx(geC()yLl_AnzW+}z{l^9ZAOHd&00JNY0w4ea zAOHd&00JN|#RT~NKi2~|BwEy{+0fv{%`%C`akqf=^{1|009sH0T2KI5C8!X009sH0T2LzM?|31 kDk;rUlamG~bxvxWR5_XBq{2y=lUYtQPD-52aH3N3Bhw>%MF0Q* literal 0 HcmV?d00001 diff --git a/.github/workflows/test.yaml b/.github/workflows/test.yaml new file mode 100644 index 0000000..5e9e653 --- /dev/null +++ b/.github/workflows/test.yaml @@ -0,0 +1,41 @@ +name: test + +on: [push,pull_request] + +jobs: + build: + + runs-on: ubuntu-latest + strategy: + matrix: + python-version: ["3.12"] + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + with: + submodules: true + + - name: Clear conda cache + run: conda clean --all -y + + - name: Conda setup + uses: conda-incubator/setup-miniconda@v4 + with: + activate-environment: tdbsumstat + environment-file: base_environment_docker.yml + auto-update-conda: true + auto-activate-base: false + python-version: ${{ matrix.python-version }} + miniforge-version: latest + + - name: Install project + shell: bash -el {0} + run: | + make dependencies_dev + make install + + - name: Run test + shell: bash -el {0} + run: | + make test-unit \ No newline at end of file diff --git a/Makefile b/Makefile index 65a14cc..f63d942 100755 --- a/Makefile +++ b/Makefile @@ -17,6 +17,12 @@ clean: find . -type d -name '__pycache__' -exec rm -rf {} + rm -rf dist build +test-all: + pytest --cov=tdbsumstat --cov-report=term-missing + +test-unit: + pytest tests/unit + dependencies: poetry install --no-root diff --git a/run_ingestion_pkgh.sh b/run_ingestion_pkgh.sh deleted file mode 100644 index dc88044..0000000 --- a/run_ingestion_pkgh.sh +++ /dev/null @@ -1,16 +0,0 @@ -#!/bin/bash -#BSUB -n 2 -#BSUB -M 4G -#BSUB -q normal -#BSUB -W 12:00 # time in HH:MM - don't put seconds! -#BSUB -G team151 -#BSUB -R "select[mem>4G] rusage[mem=4G] span[hosts=1]" -#BSUB -o tiledb_ingestion.pkgh.log # Log file for each job -#BSUB -e tiledb_ingestion.pkgh.err # Error file for each job -set -eo pipefail - -#Add the following line if you need to activate conda envs in your script -module load HGI/common/conda -source activate /software/cardinal_analysis/ht/conda_envs/tdbsumstat -module load HGI/common/nextflow/25.04.6 -nextflow run ../TileDB-sumstat/main.nf -c ingestion_config_pkgh.nf -profile sanger,conda -work-dir workdir_tiledb -resume diff --git a/tdbsumstat/utils/harmonize_ingest.py b/tdbsumstat/utils/harmonize_ingest.py index 370a2d6..1153ea7 100755 --- a/tdbsumstat/utils/harmonize_ingest.py +++ b/tdbsumstat/utils/harmonize_ingest.py @@ -34,7 +34,6 @@ def __init__(self, mapping_file: str, uri: str, type_sumstat: str, pvar_file: st self.mac = mac self.maf = maf self.permuted = permuted - print(self.permuted) def create_mapping(self): df = pd.read_csv(self.mapping_file, header=None, names=["key", "value"]) diff --git a/example_data/dummy_out_ENSG0000010000.tsv.gz b/tests/data/dummy_out_ENSG0000010000.tsv.gz similarity index 100% rename from example_data/dummy_out_ENSG0000010000.tsv.gz rename to tests/data/dummy_out_ENSG0000010000.tsv.gz diff --git a/example_data/dummy_out_ENSG0000010001.tsv.gz b/tests/data/dummy_out_ENSG0000010001.tsv.gz similarity index 100% rename from example_data/dummy_out_ENSG0000010001.tsv.gz rename to tests/data/dummy_out_ENSG0000010001.tsv.gz diff --git a/example_data/example_data_table.csv b/tests/data/example_data_table.csv similarity index 100% rename from example_data/example_data_table.csv rename to tests/data/example_data_table.csv diff --git a/example_data/locusbreaker_test_table_sc.csv b/tests/data/locusbreaker_test_table_sc.csv similarity index 100% rename from example_data/locusbreaker_test_table_sc.csv rename to tests/data/locusbreaker_test_table_sc.csv diff --git a/example_data/mapping_file_test.csv b/tests/data/mapping_file_test.csv similarity index 100% rename from example_data/mapping_file_test.csv rename to tests/data/mapping_file_test.csv diff --git a/example_data/region_list.csv b/tests/data/region_list.csv similarity index 100% rename from example_data/region_list.csv rename to tests/data/region_list.csv diff --git a/example_data/region_list_sc.csv b/tests/data/region_list_sc.csv similarity index 100% rename from example_data/region_list_sc.csv rename to tests/data/region_list_sc.csv diff --git a/example_data/snp_list_sc.csv b/tests/data/snp_list_sc.csv similarity index 100% rename from example_data/snp_list_sc.csv rename to tests/data/snp_list_sc.csv diff --git a/example_data/trait_list.csv b/tests/data/trait_list.csv similarity index 100% rename from example_data/trait_list.csv rename to tests/data/trait_list.csv diff --git a/example_data/trait_list_r.csv b/tests/data/trait_list_r.csv similarity index 100% rename from example_data/trait_list_r.csv rename to tests/data/trait_list_r.csv diff --git a/tests/unit/core/conftest.py b/tests/unit/core/conftest.py new file mode 100644 index 0000000..4a70f9a --- /dev/null +++ b/tests/unit/core/conftest.py @@ -0,0 +1,39 @@ +import pytest +import pandas as pd +from tdbsumstat.utils.harmonize_ingest import Harmonize + +@pytest.fixture +def create_df_sumstat(): + ensg00 = pd.read_csv('tests/data/dummy_out_ENSG0000010001.tsv.gz', sep = '\t') + return ensg00 + +@pytest.fixture +def create_harmonized_obj(tmp_path): + obj = Harmonize( + mapping_file= 'tests/data/mapping_file_test.csv', + uri = str(tmp_path / "test_tiledb_array"), + type_sumstat = "qtl", + pvar_file = str(tmp_path / "temp_pvar.pvar"), + type_trait = 'quant', + permuted = False, + mac = 10, + maf = 0.001, + ) + return obj + +@pytest.fixture +def create_pvar(tmp_path): + # Added list brackets to ensure pandas creates rows correctly + pvar = pd.DataFrame({ + "CHROM": ["1"], + "POS": ["1234"], + "SNPID": ["1:1234:A:G"], + "REF": ["A"], + "ALT": ["G"] + }) + + # CRITICAL: index=False prevents pandas from writing a row-number column + pvar_path = f"{tmp_path}/temp_pvar.pvar" + pvar.to_csv(pvar_path, sep='\t', index=False) + + return pvar_path \ No newline at end of file diff --git a/tests/unit/core/test_ingestion.py b/tests/unit/core/test_ingestion.py new file mode 100644 index 0000000..2fb0fec --- /dev/null +++ b/tests/unit/core/test_ingestion.py @@ -0,0 +1,40 @@ +import os +import polars as pl +from tdbsumstat.utils.harmonize_ingest import Harmonize + +def test_create_mapping(create_harmonized_obj): + # 1. Execute the method + create_harmonized_obj.create_mapping() + # 2. Check the attribute on the object + assert create_harmonized_obj.mapping_types == { + "Chr": "CHR", + "Gene": "GENE", + "cell.type": "CELL", + "pos": "POS", + "a0": "A1", + "a1": "A2", + "p": "P", + "N": "N", + "beta": "BETA", + "se": "SE", + } + +def test_create_tiledb(create_harmonized_obj, tmp_path): + create_harmonized_obj.create_tiledb() + # 2. Check the attribute on the object + assert os.path.exists(f'{tmp_path}/test_tiledb_array') + +def test_align_alleles(create_harmonized_obj, create_pvar): + create_harmonized_obj.chunk_pl = pl.DataFrame({ + "CHR": 1, + "POS": 1234, + "SNPID": "1:1234:A:G", + "BETA": -0.1, + "SE": 0.01, + "EAF": 0.3 + }) + create_harmonized_obj.align_alleles() + result_df = create_harmonized_obj.chunk_pl + assert result_df.select("BETA").item() == 0.1 + assert result_df.select("EAF").item() == 0.7 + From a8422a8ee3530f3ea76267c782d8ce17611404db Mon Sep 17 00:00:00 2001 From: bruno-ariano Date: Wed, 26 Aug 2026 12:39:05 +0200 Subject: [PATCH 2/4] Adding unit tests --- .coverage | Bin 53248 -> 0 bytes .gitignore | 1 + 2 files changed, 1 insertion(+) delete mode 100644 .coverage diff --git a/.coverage b/.coverage deleted file mode 100644 index 7b13a5879d51c52da77d41e2c35d1488b861955f..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 53248 zcmeI4%WoV-9>=>q?HSLriGnp)R+PGA9m|iw0)&7A}4;jNPB<_7ky_qD56>`;m=&v(pFJf%FQsH*ak5Q?Hy>C>c7zEx;P%Ma+cI5DD9FNN*) zY|9L7cUy!zGS3|avd0r0oujFaGp;A)dC#zWR4SC)Br))9$MpBaJ92L|$&l3+%H1$M zK_b%Lb~%SQofTgezTA?&bS)XAR-Cqb%eCdZ@75~qT_bdcg%2s2WzPz4k#Vk zNa`Ro7UbEsBG=-s60i@JcU$`Awp`Dyro+Oc4kQYljcU1l;pC*y*$)9@V}G|6FO)0o z#YOeKY7$0SYh}MGVm_WZmWr#*tlCH~BgIl*U!UEqO<;HR5EFvA^j&~`pehA+*Q+b_&dGMr5? zHaIQceQPE$`QAc8-=RUDKcz8~`OMJd%W`k+w%MoAWMJApby*sN(-}Z;#-2%;&8`=P zN&NC*5}rKw4DC$?hQ~R3%C($ir|$i#7X&s85i7 zmADj8opvA+8xGD&FAg<=J=WO1y_DFLPcA&nZ&Ihr`xosvj_LNY`HH9ERF%el?wnjt zI*3p~u|%CRpf37=W-gjtIE_%|#VOj$ZZfOg4Qwn*m?7L2FnHTt~{)0k4*gyaTKmY_l z00ck)1V8`;KmY_l00gFwKuMiZE9v_GzM|isJ_q3u2!H?xfB*=900@8p2!H?xfB*>m zIte_cmKGPX{{eV*edEdXr)va4)xT5d2O9{000@8p2!H?xfB*=900@8p2!Oye5vY|G ztJz-x`2K&Z{%1x1n;z=NTOYN$t!DG@O{e)pdbxUT$ehJm0w<(8JH2t{=JHy6M}d>vaaccT-wn(AlJin$MqI3nC{7&9D>py7{iv zw>xro;Q1jv3jNzADJ>LGG7Q_bL&G4g%MH?6ETBbiGm!zmaDl|G)Jg2b5yhg=?gyPM zd)K%bBx;*AQd=vewqyE^=i0xg*T5uZl8x7@q`Fc_HGOW}$RFZo$L9(=KGG(yR!Hqs zArt$a6$M>-zWp8Pv(x3mPDcZJh+mF6eSMZhJ3pJ5#-KmwldW%PB>PMu*{p$jk>l(c z{O$pI5TBHd5-Fc6q`WO%dZ7q$NFMt)EZOfHSs$ee&dyK;)gxBmm^SyJiz;axJEF!& z5b^#0R{aA-|ET^={e#A7J!-nGkDCjPNPoQXdGnK&)oT4R107UB00ck)1V8`;KmY_l z00b1}k`~XaNBm+xw*GHj*5ZXCN+ZjJ<`peoETT2K{%`zFi%%R??7;fJeqD>#im4s4 z{;zFn@k%k(gX{llamPp6WbUdKpDJeJ1MB}vai z{?~43@wsBk4`2V6&S`P=s1;=E|Cx(geC()yLl_AnzW+}z{l^9ZAOHd&00JNY0w4ea zAOHd&00JN|#RT~NKi2~|BwEy{+0fv{%`%C`akqf=^{1|009sH0T2KI5C8!X009sH0T2LzM?|31 kDk;rUlamG~bxvxWR5_XBq{2y=lUYtQPD-52aH3N3Bhw>%MF0Q* diff --git a/.gitignore b/.gitignore index f3d6dcd..b8fe6ac 100644 --- a/.gitignore +++ b/.gitignore @@ -3,3 +3,4 @@ dist/* work/ results/ .nextflow* +.coverage From e141e8fe21f1b5c97025adbe85a3f88fdec32389 Mon Sep 17 00:00:00 2001 From: bruno-ariano Date: Wed, 26 Aug 2026 13:58:32 +0200 Subject: [PATCH 3/4] Fix base env --- .github/workflows/test.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/test.yaml b/.github/workflows/test.yaml index 5e9e653..c35e77e 100644 --- a/.github/workflows/test.yaml +++ b/.github/workflows/test.yaml @@ -23,7 +23,7 @@ jobs: uses: conda-incubator/setup-miniconda@v4 with: activate-environment: tdbsumstat - environment-file: base_environment_docker.yml + environment-file: base_environment.yml auto-update-conda: true auto-activate-base: false python-version: ${{ matrix.python-version }} From 1a8b8ddaa8b6a237290a715676f9adf3d73b3663 Mon Sep 17 00:00:00 2001 From: bruno-ariano Date: Wed, 26 Aug 2026 14:03:16 +0200 Subject: [PATCH 4/4] Fix base env --- .github/workflows/test.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/test.yaml b/.github/workflows/test.yaml index c35e77e..87d62a2 100644 --- a/.github/workflows/test.yaml +++ b/.github/workflows/test.yaml @@ -32,7 +32,7 @@ jobs: - name: Install project shell: bash -el {0} run: | - make dependencies_dev + make dependencies make install - name: Run test