From 5b96757dbf4debf788d0cc0faeb69549ad802bbe Mon Sep 17 00:00:00 2001 From: wzl Date: Sat, 27 Jun 2026 15:00:35 +0800 Subject: [PATCH 1/8] Rename GENOS-EVEE to GENOS-VarRisk --- modules/pipeline/EXTERNAL_PATHS.md | 2 +- modules/pipeline/README.md | 10 +++---- modules/pipeline/complete_pipeline/README.md | 4 +-- .../complete_pipeline/full_pipeline_api.py | 4 +-- .../modules/genos_evee_annotation/README.md | 8 ++--- .../scripts/add_genos_evee_to_csv.py | 8 ++--- .../scripts/build_genos_evee_db.py | 4 +-- modules/pipeline/modules/hla_filter/README.md | 2 +- .../pipeline/modules/result_sorting/README.md | 4 +-- .../modules/vcf_info_to_csv/README.md | 2 +- modules/pipeline/resource_mock/README.md | 2 +- modules/pipeline/scripts/run_full_pipeline.sh | 14 ++++----- .../pipeline/test/run_mock_smoke_dryrun.sh | 2 +- modules/pixi_report/README.md | 2 +- .../examples/demo_case/output/report.md | 30 +++++++++---------- .../examples/demo_case/wide_table.csv | 2 +- modules/pixi_report/report/wide_table.py | 2 +- modules/pixi_report/scripts/export_report.py | 2 +- 18 files changed, 52 insertions(+), 52 deletions(-) diff --git a/modules/pipeline/EXTERNAL_PATHS.md b/modules/pipeline/EXTERNAL_PATHS.md index f9d6087..1925cd0 100644 --- a/modules/pipeline/EXTERNAL_PATHS.md +++ b/modules/pipeline/EXTERNAL_PATHS.md @@ -75,7 +75,7 @@ cp .env.example .env | 04 | `04_vep/raw_vep.tsv` | 原始 VEP TSV | | 04 | `04_vep/vep.log` | VEP 日志 | | 05 | `05_vcf_info_to_csv/vep_output.with_info.csv` | 合并 VCF INFO/FORMAT | -| 06 | `06_genos_evee_annotation/vep_output.with_genos_evee.csv` | GENOS-EVEE 宽表 | +| 06 | `06_genos_evee_annotation/vep_output.with_genos_evee.csv` | GENOS-VarRisk 宽表 | | 07 | `07_hla_filter/vep_output.no_hla.csv` | **最终输出**(默认 `hla_filter=yes`) | | 汇总 | `full_pipeline.outputs.tsv` | 各步路径索引 | | 日志 | `logs/full_pipeline.log` | 全流程日志 | diff --git a/modules/pipeline/README.md b/modules/pipeline/README.md index 399395c..e2e93b4 100644 --- a/modules/pipeline/README.md +++ b/modules/pipeline/README.md @@ -1,6 +1,6 @@ # OpenRare Pipeline — 罕见病变异注释与宽表输出全流程 -从患者 VCF 出发,串联 Beagle phasing、VCF 前置处理、假基因注释、VEP 多插件注释、INFO 回填、GENOS-EVEE 注释与 HLA 过滤,一键产出可解读的宽表 CSV;支持命令行与 FastAPI 两种调用方式。 +从患者 VCF 出发,串联 Beagle phasing、VCF 前置处理、假基因注释、VEP 多插件注释、INFO 回填、GENOS-VarRisk 注释与 HLA 过滤,一键产出可解读的宽表 CSV;支持命令行与 FastAPI 两种调用方式。 ## 目录 @@ -123,7 +123,7 @@ full_log /path/to/output/logs/full_pipeline.log 最终 CSV 表头示例(列较多,此处仅示意): ```text -#CHROM,POS,REF,ALT,...,Gene,Consequence,CLIN_SIG,...,GENOS-EVEE,... +#CHROM,POS,REF,ALT,...,Gene,Consequence,CLIN_SIG,...,GENOS-VarRisk,... chr1,1197557,G,A,...,TTLL10,missense_variant,...,0.42,... ``` @@ -315,7 +315,7 @@ pixi run bash scripts/run_full_pipeline.sh --help | 03 | `03_pseudogene_annotation/` | 假基因 INFO 注释(可跳过) | `preprocessed.pseudogene_annotated.vcf.gz` | | 04 | `04_vep/` | VEP + CADD/SpliceAI/dbNSFP/LoFTEE/ClinVar 等 | `vep_output.base.csv` | | 05 | `05_vcf_info_to_csv/` | VCF INFO/FORMAT 回填到 CSV | `vep_output.with_info.csv` | -| 06 | `06_genos_evee_annotation/` | GENOS-EVEE 疾病预测分数 | `vep_output.with_genos_evee.csv` | +| 06 | `06_genos_evee_annotation/` | GENOS-VarRisk 疾病预测分数 | `vep_output.with_genos_evee.csv` | | 07 | `07_hla_filter/` | 删除 GRCh38 HLA/MHC 区行(可 `--hla-filter no` 跳过) | **`vep_output.no_hla.csv`** | --- @@ -338,7 +338,7 @@ pixi run bash scripts/run_full_pipeline.sh --help |------|------| | `/04_vep/vep_output.base.csv` | VEP 基础 CSV | | `/04_vep/raw_vep.tsv` | VEP 原始 TSV(默认保留) | -| `/06_genos_evee_annotation/vep_output.with_genos_evee.csv` | GENOS-EVEE 宽表(HLA 过滤前) | +| `/06_genos_evee_annotation/vep_output.with_genos_evee.csv` | GENOS-VarRisk 宽表(HLA 过滤前) | | `/07_hla_filter/vep_output.no_hla.csv` | **最终宽表**(默认) | | `/full_pipeline.outputs.tsv` | 各步产物路径索引 | | `/logs/full_pipeline.log` | 全流程日志 | @@ -407,7 +407,7 @@ VEP 配置 [`modules/vep_runner/config/vep_runner_config.json`](modules/vep_runn |------|------| | [complete_pipeline/README.md](complete_pipeline/README.md) | API 与 CLI 详细参数 | | [modules/vep_runner/README.md](modules/vep_runner/README.md) | VEP 插件与注释库安装 | -| [modules/genos_evee_annotation/README.md](modules/genos_evee_annotation/README.md) | GENOS-EVEE 注释与数据库构建 | +| [modules/genos_evee_annotation/README.md](modules/genos_evee_annotation/README.md) | GENOS-VarRisk 注释与数据库构建 | | [modules/hla_filter/README.md](modules/hla_filter/README.md) | HLA/MHC 区行过滤 | | [modules/result_sorting/README.md](modules/result_sorting/README.md) | 可选致病性排序(主流程默认不启用) | | [resource_mock/README.md](resource_mock/README.md) | 联调用迷你 mock 库 | diff --git a/modules/pipeline/complete_pipeline/README.md b/modules/pipeline/complete_pipeline/README.md index 325d08f..d0631f7 100644 --- a/modules/pipeline/complete_pipeline/README.md +++ b/modules/pipeline/complete_pipeline/README.md @@ -3,7 +3,7 @@ > 完整架构和数据资源下载说明见上一层 `../README.md`。本文件保留主程序/API 的详细调用说明。 -本目录包含 FastAPI 服务(`full_pipeline_api.py`)与 API 任务目录,负责把 liftover(可选)、phasing、VCF 前处理、假基因注释、VEP runner、VCF INFO 回填、GENOS-EVEE 注释、HLA 过滤串成一个完整流程。 +本目录包含 FastAPI 服务(`full_pipeline_api.py`)与 API 任务目录,负责把 liftover(可选)、phasing、VCF 前处理、假基因注释、VEP runner、VCF INFO 回填、GENOS-VarRisk 注释、HLA 过滤串成一个完整流程。 主程序: @@ -307,7 +307,7 @@ API 的字段和主程序参数一一对应。常规字段和高级覆盖字段 | `top_k_transcripts` | `--top-k-transcripts` | 高级覆盖 | 每个 variant-gene 保留的转录本数量。 | | `clinical_tissue` | `--clinical-tissue` | 高级覆盖 | 手动传 GTEx tissue。 | | `keep_raw_vep` | `--keep-raw-vep` | 高级覆盖 | `true` 对应 `yes`,`false` 对应 `no`。 | -| `genos_evee_db` | `--genos-evee-db` | 高级覆盖 | GENOS-EVEE CPRA 数据库路径。 | +| `genos_evee_db` | `--GENOS-VarRisk-db` | 高级覆盖 | GENOS-VarRisk CPRA 数据库路径。 | | `dry_run` | `--dry-run` | 调试 | 只打印命令,不实际运行。 | 示例:API 只跑 chr1: diff --git a/modules/pipeline/complete_pipeline/full_pipeline_api.py b/modules/pipeline/complete_pipeline/full_pipeline_api.py index f528c35..a1bdc80 100644 --- a/modules/pipeline/complete_pipeline/full_pipeline_api.py +++ b/modules/pipeline/complete_pipeline/full_pipeline_api.py @@ -75,7 +75,7 @@ class RunRequest(BaseModel): java_bin: Optional[str] = Field(None, description="Advanced override: Java executable for Beagle") top_k_transcripts: Optional[int] = Field(None, description="Advanced override: transcript selection count") clinical_tissue: str = Field("", description="Advanced override: GTEx tissue name for phenotype-aware transcript expression") - genos_evee_db: Optional[str] = Field(None, description="Advanced override: indexed GENOS-EVEE CPRA TSV.GZ") + genos_evee_db: Optional[str] = Field(None, description="Advanced override: indexed GENOS-VarRisk CPRA TSV.GZ") keep_raw_vep: Optional[bool] = Field(None, description="Advanced override: keep raw VEP TSV") dry_run: bool = False @@ -298,7 +298,7 @@ def add_option(name: str, value) -> None: add_option("--java-bin", resolve_path(req.java_bin) if req.java_bin else None) add_option("--top-k-transcripts", req.top_k_transcripts) add_option("--clinical-tissue", req.clinical_tissue) - add_option("--genos-evee-db", resolve_path(req.genos_evee_db) if req.genos_evee_db else None) + add_option("--GENOS-VarRisk-db", resolve_path(req.genos_evee_db) if req.genos_evee_db else None) if req.keep_raw_vep is not None: cmd.extend(["--keep-raw-vep", "yes" if req.keep_raw_vep else "no"]) if req.dry_run: diff --git a/modules/pipeline/modules/genos_evee_annotation/README.md b/modules/pipeline/modules/genos_evee_annotation/README.md index 0ecca36..7a367e3 100644 --- a/modules/pipeline/modules/genos_evee_annotation/README.md +++ b/modules/pipeline/modules/genos_evee_annotation/README.md @@ -1,13 +1,13 @@ -# GENOS-EVEE 疾病预测数据库注释模块 +# GENOS-VarRisk 疾病预测数据库注释模块 -将疾病预测模型输出中的 `p_fusion` 按 CPRA 匹配到宽表,列名固定为 `GENOS-EVEE`。对应主流程第 **06** 步。 +将疾病预测模型输出中的 `p_fusion` 按 CPRA 匹配到宽表,列名固定为 `GENOS-VarRisk`。对应主流程第 **06** 步。 ## 脚本 | 脚本 | 用途 | |------|------| | `scripts/build_genos_evee_db.py` | 将 prediction TSV shard 构建为 bgzip + tabix 数据库 | -| `scripts/add_genos_evee_to_csv.py` | 在 05 宽表中加入 `GENOS-EVEE` 列 | +| `scripts/add_genos_evee_to_csv.py` | 在 05 宽表中加入 `GENOS-VarRisk` 列 | ## 主流程 @@ -15,7 +15,7 @@ bash scripts/run_full_pipeline.sh \ --input-vcf /path/to/input.vcf.gz \ --out-dir /path/to/output \ - --genos-evee-db /path/to/genos_evee.cpra.tsv.gz + --GENOS-VarRisk-db /path/to/genos_evee.cpra.tsv.gz ``` 默认数据库路径由 `.env` 的 `FULL_PIPELINE_GENOS_EVEE_DB` 或 `config/path_utils.py` 解析。数据库不存在时流程仍可运行,该列全部填 `-`。 diff --git a/modules/pipeline/modules/genos_evee_annotation/scripts/add_genos_evee_to_csv.py b/modules/pipeline/modules/genos_evee_annotation/scripts/add_genos_evee_to_csv.py index 75cd204..b2fd092 100644 --- a/modules/pipeline/modules/genos_evee_annotation/scripts/add_genos_evee_to_csv.py +++ b/modules/pipeline/modules/genos_evee_annotation/scripts/add_genos_evee_to_csv.py @@ -9,7 +9,7 @@ import tempfile from pathlib import Path -OUTPUT_COLUMN = "GENOS-EVEE" +OUTPUT_COLUMN = "GENOS-VarRisk" def normalize_chrom(value: str) -> str: @@ -70,7 +70,7 @@ def annotate(input_csv: Path, database: Path | None, output_csv: Path, log_json: output_csv.parent.mkdir(parents=True, exist_ok=True) database_available = database is not None and database.is_file() and Path(f"{database}.tbi").is_file() if database is not None and database.exists() and not database_available: - raise FileNotFoundError(f"GENOS-EVEE database index not found: {database}.tbi") + raise FileNotFoundError(f"GENOS-VarRisk database index not found: {database}.tbi") if database_available and shutil.which("tabix") is None: raise RuntimeError("tabix command not found") @@ -153,9 +153,9 @@ def annotate(input_csv: Path, database: Path | None, output_csv: Path, log_json: def main() -> None: - parser = argparse.ArgumentParser(description="Add GENOS-EVEE p_fusion scores to a V3 wide CSV before sorting.") + parser = argparse.ArgumentParser(description="Add GENOS-VarRisk p_fusion scores to a V3 wide CSV before sorting.") parser.add_argument("--input-csv", required=True) - parser.add_argument("--database", help="Indexed GENOS-EVEE .tsv.gz; missing/empty means fill '-'") + parser.add_argument("--database", help="Indexed GENOS-VarRisk .tsv.gz; missing/empty means fill '-'") parser.add_argument("--output-csv", required=True) parser.add_argument("--log-json") args = parser.parse_args() diff --git a/modules/pipeline/modules/genos_evee_annotation/scripts/build_genos_evee_db.py b/modules/pipeline/modules/genos_evee_annotation/scripts/build_genos_evee_db.py index 434e261..9d21650 100644 --- a/modules/pipeline/modules/genos_evee_annotation/scripts/build_genos_evee_db.py +++ b/modules/pipeline/modules/genos_evee_annotation/scripts/build_genos_evee_db.py @@ -165,7 +165,7 @@ def build_database(inputs: list[Path], output: Path, threads: int, log_json: Pat ) unique = 0 with sorted_path.open("r", encoding="utf-8") as src, merged.open("w", encoding="utf-8") as dst: - dst.write("#CHROM\tPOS\tREF\tALT\tGENOS-EVEE\n") + dst.write("#CHROM\tPOS\tREF\tALT\tGENOS-VarRisk\n") previous: tuple[str, str, str, str] | None = None for line in src: parts = line.rstrip("\n").split("\t") @@ -190,7 +190,7 @@ def build_database(inputs: list[Path], output: Path, threads: int, log_json: Pat def main() -> None: - parser = argparse.ArgumentParser(description="Build indexed GENOS-EVEE CPRA database from prediction TSV shards.") + parser = argparse.ArgumentParser(description="Build indexed GENOS-VarRisk CPRA database from prediction TSV shards.") parser.add_argument("--input", action="append", required=True, help="TSV file, directory, or glob; repeatable") parser.add_argument("--output", required=True, help="Output .tsv.gz database") parser.add_argument("--threads", type=int, default=4) diff --git a/modules/pipeline/modules/hla_filter/README.md b/modules/pipeline/modules/hla_filter/README.md index 1618c3a..169c2c5 100644 --- a/modules/pipeline/modules/hla_filter/README.md +++ b/modules/pipeline/modules/hla_filter/README.md @@ -1,6 +1,6 @@ # HLA/MHC 区域行过滤模块 -从 GENOS-EVEE 宽表中删除 GRCh38 HLA/MHC 区域的变异行。对应主流程第 **07** 步(默认开启)。 +从 GENOS-VarRisk 宽表中删除 GRCh38 HLA/MHC 区域的变异行。对应主流程第 **07** 步(默认开启)。 ## 过滤区间 diff --git a/modules/pipeline/modules/result_sorting/README.md b/modules/pipeline/modules/result_sorting/README.md index db4a02b..6f05bea 100644 --- a/modules/pipeline/modules/result_sorting/README.md +++ b/modules/pipeline/modules/result_sorting/README.md @@ -4,7 +4,7 @@ ## 与主流程的关系 -当前 `scripts/run_full_pipeline.sh` **默认不调用**本模块。主流程在 06 GENOS-EVEE 注释之后,由 07 HLA 过滤产出最终宽表: +当前 `scripts/run_full_pipeline.sh` **默认不调用**本模块。主流程在 06 GENOS-VarRisk 注释之后,由 07 HLA 过滤产出最终宽表: ```text 05 vcf_info_to_csv → 06 genos_evee_annotation → 07 hla_filter(默认最终 CSV) @@ -14,7 +14,7 @@ ## 输入 -- `--input-csv`:带 `vcf_info_*`(及可选 `GENOS-EVEE`)列的宽表,通常来自 `05_vcf_info_to_csv/` 或 `06_genos_evee_annotation/`。 +- `--input-csv`:带 `vcf_info_*`(及可选 `GENOS-VarRisk`)列的宽表,通常来自 `05_vcf_info_to_csv/` 或 `06_genos_evee_annotation/`。 - `--output-csv`:排序后的 CSV。 ## 单独运行示例 diff --git a/modules/pipeline/modules/vcf_info_to_csv/README.md b/modules/pipeline/modules/vcf_info_to_csv/README.md index acfa6b2..4647b2b 100644 --- a/modules/pipeline/modules/vcf_info_to_csv/README.md +++ b/modules/pipeline/modules/vcf_info_to_csv/README.md @@ -2,7 +2,7 @@ 功能:读取进入 04 VEP runner 的 VCF,把 VCF 中所有 INFO 字段展开为 `vcf_info_` 列,并把样本 FORMAT 展开为 `vcf_format_*` 列,追加到 04 生成的 VEP CSV 中。 -主流程中本模块为第 **05** 步;输出 `vep_output.with_info.csv` 进入 **06** GENOS-EVEE 注释,再经 **07** HLA 过滤(默认)得到最终宽表。 +主流程中本模块为第 **05** 步;输出 `vep_output.with_info.csv` 进入 **06** GENOS-VarRisk 注释,再经 **07** HLA 过滤(默认)得到最终宽表。 ## 输入 diff --git a/modules/pipeline/resource_mock/README.md b/modules/pipeline/resource_mock/README.md index 634a31a..a2ff706 100644 --- a/modules/pipeline/resource_mock/README.md +++ b/modules/pipeline/resource_mock/README.md @@ -25,7 +25,7 @@ bash scripts/run_full_pipeline.sh \ --phasing no \ --ccre-bed resource_mock/regulatory/hg38/encode_screen_v4_grch38_ccre.slim.mock.bed.gz \ --ncrna-bed resource_mock/ncrna/hg38/gencode.v49.ncrna_gene.slim.mock.bed.gz \ - --genos-evee-db resource_mock/genos_evee/genos_evee.cpra.mock.tsv.gz \ + --GENOS-VarRisk-db resource_mock/genos_evee/genos_evee.cpra.mock.tsv.gz \ --dry-run ``` diff --git a/modules/pipeline/scripts/run_full_pipeline.sh b/modules/pipeline/scripts/run_full_pipeline.sh index b6eafcc..b20f3a3 100755 --- a/modules/pipeline/scripts/run_full_pipeline.sh +++ b/modules/pipeline/scripts/run_full_pipeline.sh @@ -49,7 +49,7 @@ Advanced optional overrides, usually not needed: --java-bin PATH Java executable for Beagle --top-k-transcripts N VEP transcript selection count, default: 5 --clinical-tissue NAME Optional clinical tissue for VEP runner - --genos-evee-db FILE Indexed GENOS-EVEE CPRA TSV.GZ; default: resource/genos_evee/genos_evee.cpra.tsv.gz + --GENOS-VarRisk-db FILE Indexed GENOS-VarRisk CPRA TSV.GZ; default: resource/genos_evee/genos_evee.cpra.tsv.gz --keep-raw-vep yes|no Keep raw VEP TSV, default: yes --input-assembly SPEC Input assembly: auto, GRCh37, or GRCh38 (default: auto) --dry-run Print commands only @@ -107,7 +107,7 @@ while [[ $# -gt 0 ]]; do --pseudogene-annotation) PSEUDOGENE_ANNOTATION="${2:?}"; shift 2 ;; --hla-filter) HLA_FILTER="${2:?}"; shift 2 ;; --clinical-tissue) CLINICAL_TISSUE="${2:?}"; shift 2 ;; - --genos-evee-db) GENOS_EVEE_DB="${2:?}"; shift 2 ;; + --GENOS-VarRisk-db) GENOS_EVEE_DB="${2:?}"; shift 2 ;; --keep-raw-vep) KEEP_RAW_VEP="${2:?}"; shift 2 ;; --input-assembly) INPUT_ASSEMBLY="${2:?}"; shift 2 ;; --dry-run) DRY_RUN=yes; shift ;; @@ -152,7 +152,7 @@ fi [[ -s "$PSEUDOGENE_PY" ]] || { echo "ERROR: pseudogene script not found: $PSEUDOGENE_PY" >&2; exit 1; } [[ -s "$INFO_TO_CSV_SCRIPT" ]] || { echo "ERROR: INFO-to-CSV script not found: $INFO_TO_CSV_SCRIPT" >&2; exit 1; } [[ -s "$ENSURE_HEADERS_SCRIPT" ]] || { echo "ERROR: ensure-header script not found: $ENSURE_HEADERS_SCRIPT" >&2; exit 1; } -[[ -s "$GENOS_EVEE_SCRIPT" ]] || { echo "ERROR: GENOS-EVEE annotation script not found: $GENOS_EVEE_SCRIPT" >&2; exit 1; } +[[ -s "$GENOS_EVEE_SCRIPT" ]] || { echo "ERROR: GENOS-VarRisk annotation script not found: $GENOS_EVEE_SCRIPT" >&2; exit 1; } [[ -s "$HLA_FILTER_SCRIPT" ]] || { echo "ERROR: HLA filter script not found: $HLA_FILTER_SCRIPT" >&2; exit 1; } case "$INPUT_ASSEMBLY" in auto|GRCh37|GRCh38) ;; @@ -350,7 +350,7 @@ vep_evee_csv="${OUT_DIR}/06_genos_evee_annotation/vep_output.with_genos_evee.csv vep_evee_log="${OUT_DIR}/06_genos_evee_annotation/genos_evee_annotation.log.json" hla_filtered_csv="${OUT_DIR}/07_hla_filter/vep_output.no_hla.csv" hla_filter_log="${OUT_DIR}/07_hla_filter/hla_filter.log.json" -# Final sorting is intentionally disabled. The HLA-filtered GENOS-EVEE wide table is the final CSV when HLA_FILTER=yes. +# Final sorting is intentionally disabled. The HLA-filtered GENOS-VarRisk wide table is the final CSV when HLA_FILTER=yes. vep_csv="$vep_evee_csv" vep_cmd=(python3 "$VEP_SCRIPT" -i "$pseudo_vcf" -o "$vep_base_csv" --config "$VEP_CONFIG" --format vcf --hgvs --fork "$FORK" --top-k-transcripts "$TOP_K_TRANSCRIPTS" --no-pseudogene-annotation --no-regulatory-annotation --no-vcf-info-to-csv --no-pathogenic-ranking --log "$vep_log") if [[ "$KEEP_RAW_VEP" == yes ]]; then @@ -379,10 +379,10 @@ genos_evee_cmd=(python3 "$GENOS_EVEE_SCRIPT" \ --output-csv "$vep_evee_csv" \ --log-json "$vep_evee_log") if [[ -s "$GENOS_EVEE_DB" ]]; then - [[ -s "${GENOS_EVEE_DB}.tbi" ]] || { echo "ERROR: GENOS-EVEE database index not found: ${GENOS_EVEE_DB}.tbi" >&2; exit 1; } + [[ -s "${GENOS_EVEE_DB}.tbi" ]] || { echo "ERROR: GENOS-VarRisk database index not found: ${GENOS_EVEE_DB}.tbi" >&2; exit 1; } genos_evee_cmd+=(--database "$GENOS_EVEE_DB") else - echo "WARNING: GENOS-EVEE database not found; GENOS-EVEE column will be filled with '-': $GENOS_EVEE_DB" + echo "WARNING: GENOS-VarRisk database not found; GENOS-VarRisk column will be filled with '-': $GENOS_EVEE_DB" fi run_cmd "${genos_evee_cmd[@]}" @@ -390,7 +390,7 @@ if [[ "$HLA_FILTER" == yes ]]; then run_cmd python3 "$HLA_FILTER_SCRIPT" --input-csv "$vep_evee_csv" --output-csv "$hla_filtered_csv" --log-json "$hla_filter_log" vep_csv="$hla_filtered_csv" else - printf '[%s] SKIP HLA filter: keeping GENOS-EVEE wide CSV: %s\n' "$(date '+%F %T')" "$vep_csv" + printf '[%s] SKIP HLA filter: keeping GENOS-VarRisk wide CSV: %s\n' "$(date '+%F %T')" "$vep_csv" fi { diff --git a/modules/pipeline/test/run_mock_smoke_dryrun.sh b/modules/pipeline/test/run_mock_smoke_dryrun.sh index 31a609b..fd0fdd5 100755 --- a/modules/pipeline/test/run_mock_smoke_dryrun.sh +++ b/modules/pipeline/test/run_mock_smoke_dryrun.sh @@ -31,7 +31,7 @@ bash "${PIPELINE_ROOT}/scripts/run_full_pipeline.sh" \ --hla-filter yes \ --ccre-bed "${MOCK_ROOT}/regulatory/hg38/encode_screen_v4_grch38_ccre.slim.mock.bed.gz" \ --ncrna-bed "${MOCK_ROOT}/ncrna/hg38/gencode.v49.ncrna_gene.slim.mock.bed.gz" \ - --genos-evee-db "${MOCK_ROOT}/genos_evee/genos_evee.cpra.mock.tsv.gz" \ + --GENOS-VarRisk-db "${MOCK_ROOT}/genos_evee/genos_evee.cpra.mock.tsv.gz" \ --dry-run log "OK: mock smoke dry-run completed" diff --git a/modules/pixi_report/README.md b/modules/pixi_report/README.md index c67a61c..84a27b1 100644 --- a/modules/pixi_report/README.md +++ b/modules/pixi_report/README.md @@ -58,7 +58,7 @@ FastAPI(`POST /report/stream`)需要三个输入文件。路径可为绝对 | `clinical_best_tissue` / `gtex_transcript_top5_tissues` | 表达组织 | | `evidence_summary` | 证据摘要 | | `ppi_final` | 变异级 PPI 得分(Gene Card 展示回退来源) | -| `GENOS-EVEE` | Genos-Mutation 评分(可选;宽表列名仍为 `GENOS-EVEE`,报告中展示为 Genos-Mutation) | +| `GENOS-VarRisk` | Genos-Mutation 评分(可选;宽表列名仍为 `GENOS-VarRisk`,报告中展示为 Genos-Mutation) | 完整列表示例见 `examples/demo_case/wide_table.csv` 或 `fixtures/wide_table.csv`。 diff --git a/modules/pixi_report/examples/demo_case/output/report.md b/modules/pixi_report/examples/demo_case/output/report.md index ef90e66..e556aea 100644 --- a/modules/pixi_report/examples/demo_case/output/report.md +++ b/modules/pixi_report/examples/demo_case/output/report.md @@ -79,11 +79,11 @@ evidence_score: -32 | 主要关联通路 | - | | 致病性排名 | #1 | | PPI 得分 | 0.6265 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | #### 3.1.2 变异列表 -| 变异 | 转录本 / 后果 | GENOS-EVEE | 评分 | ClinVar(VAF) | +| 变异 | 转录本 / 后果 | GENOS-VarRisk | 评分 | ClinVar(VAF) | |------|---------------|------------|------|----------------| | HLA-DPA1 p.Thr259Pro | ENST00000692443.1 · missense_variant | - | CADD 3.450 | -(VAF 100.0%) | @@ -114,7 +114,7 @@ evidence_score: -32 |------|------| | 转录本 | ENST00000692443.1 | | RefSeq | NM_033554.4,NM_001405020.1 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | | HGVSc | c.775A>C | | HGVSp | p.Thr259Pro | | VEP 后果 | missense_variant | @@ -199,11 +199,11 @@ evidence_score: -32 | 主要关联通路 | - | | 致病性排名 | #2 | | PPI 得分 | 0.4378 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | #### 3.2.2 变异列表 -| 变异 | 转录本 / 后果 | GENOS-EVEE | 评分 | ClinVar(VAF) | +| 变异 | 转录本 / 后果 | GENOS-VarRisk | 评分 | ClinVar(VAF) | |------|---------------|------------|------|----------------| | CXCL17 p.Glu45Asp | ENST00000601181.6 · missense_variant | - | CADD 13.65 | -(VAF 61.8%) | @@ -237,7 +237,7 @@ evidence_score: -32 |------|------| | 转录本 | ENST00000601181.6 | | RefSeq | NM_198477.3 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | | HGVSc | c.135A>C | | HGVSp | p.Glu45Asp | | VEP 后果 | missense_variant | @@ -324,11 +324,11 @@ evidence_score: +0 | 主要关联通路 | - | | 致病性排名 | #4 | | PPI 得分 | 0.4574 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | #### 3.3.2 变异列表 -| 变异 | 转录本 / 后果 | GENOS-EVEE | 评分 | ClinVar(VAF) | +| 变异 | 转录本 / 后果 | GENOS-VarRisk | 评分 | ClinVar(VAF) | |------|---------------|------------|------|----------------| | DDI2 chr1:15660052 T>C | ENST00000480945.6 · 3_prime_UTR_variant | - | CADD 16.17 | -(VAF 66.7%) | @@ -362,7 +362,7 @@ evidence_score: +0 |------|------| | 转录本 | ENST00000480945.6 | | RefSeq | NM_032341.5 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | | HGVSc | c.*262T>C | | HGVSp | - | | VEP 后果 | 3_prime_UTR_variant | @@ -447,11 +447,11 @@ evidence_score: -36 | 主要关联通路 | - | | 致病性排名 | #5 | | PPI 得分 | 0.5393 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | #### 3.4.2 变异列表 -| 变异 | 转录本 / 后果 | GENOS-EVEE | 评分 | ClinVar(VAF) | +| 变异 | 转录本 / 后果 | GENOS-VarRisk | 评分 | ClinVar(VAF) | |------|---------------|------------|------|----------------| | BMP8B chr1:39769817 T>C | ENST00000372827.8 · intron_variant | - | CADD 9.907 | -(VAF 34.5%) | @@ -484,7 +484,7 @@ evidence_score: -36 |------|------| | 转录本 | ENST00000372827.8 | | RefSeq | NM_001720.5 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | | HGVSc | c.673+4491A>G | | HGVSp | - | | VEP 后果 | intron_variant | @@ -568,11 +568,11 @@ evidence_score: -38 | 主要关联通路 | - | | 致病性排名 | #8 | | PPI 得分 | 0.5175 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | #### 3.5.2 变异列表 -| 变异 | 转录本 / 后果 | GENOS-EVEE | 评分 | ClinVar(VAF) | +| 变异 | 转录本 / 后果 | GENOS-VarRisk | 评分 | ClinVar(VAF) | |------|---------------|------------|------|----------------| | TAS2R19 p.Met297Val | ENST00000390673.2 · missense_variant | - | CADD 0.002 | -(VAF 49.6%) | @@ -598,7 +598,7 @@ evidence_score: -38 |------|------| | 转录本 | ENST00000390673.2 | | RefSeq | NM_176888.2 | -| GENOS-EVEE | - | +| GENOS-VarRisk | - | | HGVSc | c.889A>G | | HGVSp | p.Met297Val | | VEP 后果 | missense_variant | diff --git a/modules/pixi_report/examples/demo_case/wide_table.csv b/modules/pixi_report/examples/demo_case/wide_table.csv index 9a741ad..3d2273c 100644 --- a/modules/pixi_report/examples/demo_case/wide_table.csv +++ b/modules/pixi_report/examples/demo_case/wide_table.csv @@ -1,4 +1,4 @@ -chrom,pos,ref,alt,GENOS-EVEE,vcf_info_AC,vcf_info_AF,vcf_info_AN,vcf_info_BaseQRankSum,vcf_info_DP,vcf_info_END,vcf_info_ExcessHet,vcf_info_FS,vcf_info_InbreedingCoeff,vcf_info_MLEAC,vcf_info_MLEAF,vcf_info_MQ,vcf_info_MQRankSum,vcf_info_QD,vcf_info_RAW_MQandDP,vcf_info_ReadPosRankSum,vcf_info_SOR,vcf_info_BEAGLE_PHASED,vcf_info_CHN_REF_SUPPORT,vcf_info_CHN_ALT_CARRIER_COUNT,vcf_info_CHN_ALT_AC,vcf_info_PHASING_CONFIDENCE,vcf_info_VAF,vcf_info_REF_DP,vcf_info_ALT_DP,vcf_info_REG_CCRE_ID,vcf_info_REG_CCRE_CLASS,vcf_info_REG_CCRE_COUNT,vcf_info_REG_CCRE_SOURCE,vcf_info_NCRNA_GENE_ID,vcf_info_NCRNA_GENE_NAME,vcf_info_NCRNA_GENE_TYPE,vcf_info_NCRNA_GENE_COUNT,vcf_info_NCRNA_SOURCE,vcf_info_is_pseudogene,vcf_info_pseudogene_name,vcf_info_pseudogene_source,vcf_format_AD,vcf_format_DP,vcf_format_GQ,vcf_format_GT,vcf_format_MIN_DP,vcf_format_PGT,vcf_format_PID,vcf_format_PL,vcf_format_PS,vcf_format_SB,vcf_format_RGQ,gene_symbol,all_genes,transcript_id,refseq_id,biotype,canonical,mane,mane_select,mane_plus_clinical,appris,tsl,ccds,vep_pick,transcript_flags,consequence,impact,hgvsc,hgvsp,cdna_position,cds_position,protein_position,amino_acids,codons,exon,intron,strand,protein_domains,revel_score,cadd_phred,gnomAD_popmax_AF,gnomAD_eas_AF,gnomAD_nhomalt,spliceAI_ds_max,spliceAI_type,loftee_lof_flag,loftee_lof_filter,clinvar_significance,clinvar_review_status,clinvar_star_rating,clinical_gtex_tissue_whitelist,clinical_best_tissue,clinical_transcript_tpm,gtex_transcript_max_tissue,gtex_transcript_max_tpm,gtex_transcript_top5_tissues,gtex_max_tissue_in_clinical_whitelist,clinical_vs_global_tpm_ratio,gtex_lookup_status,tx_consequence_score,tx_confidence_score,clinical_expression_score,tx_tie_breaker_score,tx_selection_score,tx_rank_within_variant,tx_eligibility,tx_exclusion_reason,tx_rescue_reason,tx_selected_reason,pathogenic_rank,evidence_summary,_vkey,evolve_score,evolve_rank,gene_score,ppi_final,pathogenic_score,pathogenic_rank_1 +chrom,pos,ref,alt,GENOS-VarRisk,vcf_info_AC,vcf_info_AF,vcf_info_AN,vcf_info_BaseQRankSum,vcf_info_DP,vcf_info_END,vcf_info_ExcessHet,vcf_info_FS,vcf_info_InbreedingCoeff,vcf_info_MLEAC,vcf_info_MLEAF,vcf_info_MQ,vcf_info_MQRankSum,vcf_info_QD,vcf_info_RAW_MQandDP,vcf_info_ReadPosRankSum,vcf_info_SOR,vcf_info_BEAGLE_PHASED,vcf_info_CHN_REF_SUPPORT,vcf_info_CHN_ALT_CARRIER_COUNT,vcf_info_CHN_ALT_AC,vcf_info_PHASING_CONFIDENCE,vcf_info_VAF,vcf_info_REF_DP,vcf_info_ALT_DP,vcf_info_REG_CCRE_ID,vcf_info_REG_CCRE_CLASS,vcf_info_REG_CCRE_COUNT,vcf_info_REG_CCRE_SOURCE,vcf_info_NCRNA_GENE_ID,vcf_info_NCRNA_GENE_NAME,vcf_info_NCRNA_GENE_TYPE,vcf_info_NCRNA_GENE_COUNT,vcf_info_NCRNA_SOURCE,vcf_info_is_pseudogene,vcf_info_pseudogene_name,vcf_info_pseudogene_source,vcf_format_AD,vcf_format_DP,vcf_format_GQ,vcf_format_GT,vcf_format_MIN_DP,vcf_format_PGT,vcf_format_PID,vcf_format_PL,vcf_format_PS,vcf_format_SB,vcf_format_RGQ,gene_symbol,all_genes,transcript_id,refseq_id,biotype,canonical,mane,mane_select,mane_plus_clinical,appris,tsl,ccds,vep_pick,transcript_flags,consequence,impact,hgvsc,hgvsp,cdna_position,cds_position,protein_position,amino_acids,codons,exon,intron,strand,protein_domains,revel_score,cadd_phred,gnomAD_popmax_AF,gnomAD_eas_AF,gnomAD_nhomalt,spliceAI_ds_max,spliceAI_type,loftee_lof_flag,loftee_lof_filter,clinvar_significance,clinvar_review_status,clinvar_star_rating,clinical_gtex_tissue_whitelist,clinical_best_tissue,clinical_transcript_tpm,gtex_transcript_max_tissue,gtex_transcript_max_tpm,gtex_transcript_top5_tissues,gtex_max_tissue_in_clinical_whitelist,clinical_vs_global_tpm_ratio,gtex_lookup_status,tx_consequence_score,tx_confidence_score,clinical_expression_score,tx_tie_breaker_score,tx_selection_score,tx_rank_within_variant,tx_eligibility,tx_exclusion_reason,tx_rescue_reason,tx_selected_reason,pathogenic_rank,evidence_summary,_vkey,evolve_score,evolve_rank,gene_score,ppi_final,pathogenic_score,pathogenic_rank_1 chr6,33068658,T,G,-,2,1,2,-,50,-,0,0,-,2,1,60,-,33.19,-,-,0.963,1,POLYMORPHIC,337,510,HIGH,1,0,48,EH38E3701776,dELS,1,ENCODE_SCREEN_v4_GRCh38,-,-,-,-,-,No,-,-,"0,48",48,99,1|1,-,-,-,"1607,144,0",-,-,-,HLA-DPA1,HLA-DPA1,ENST00000692443.1,"NM_033554.4,NM_001405020.1",protein_coding,YES,MANE_Select,NM_033554.4,-,P1,-,CCDS4764.1,1,-,missense_variant,MODERATE,c.775A>C,p.Thr259Pro,854,775,259,T/P,Acc/Ccc,4/5,-,-1,"AFDB-ENSP_mappings:AF-P20036-F1,Phobius:CYTOPLASMIC_DOMAIN",-,3.450,6.409020e-01,6.409020e-01,47124,0,-,-,-,-,-,0,"Nerve - Tibial,Brain - Spinal cord (cervical c-1)",Nerve - Tibial,129.8,Cells - EBV-transformed lymphocytes,573.9,Cells - EBV-transformed lymphocytes:573.9;Small Intestine - Terminal Ileum - Lymphoid Aggregate:327.7;Spleen:280;Lung:259.6;Liver - Portal Tract:189,NO,0.2261,ok,22,35,20,5,82,1,selected,-,-,top_k,-,consequence=missense_variant(+15); EAS_AF=6.409020e-01; frequency(-50); domain(+3); total=-32,6:33068658:T:G,16.998,12,0.460094,0.6264820254272677,51.10769827749661,1 chr19,42433801,T,G,-,1,0.5,2,-2.344,56,-,0,2.341,-,1,0.5,60,0,14.27,-,0.338,1.048,1,POLYMORPHIC,20,20,HIGH,0.618182,21,34,EH38E3307776,dELS,1,ENCODE_SCREEN_v4_GRCh38,ENSG00000213904.12,LIPE-AS1,lncRNA,1,GENCODE_v49_GRCh38_ncRNA_gene,No,-,-,"21,34",55,99,0|1,-,-,-,"792,0,498",-,-,-,CXCL17,"CXCL17,LIPE-AS1",ENST00000601181.6,NM_198477.3,protein_coding,YES,MANE_Select,NM_198477.3,-,P1,1,CCDS12608.1,1,-,missense_variant,MODERATE,c.135A>C,p.Glu45Asp,249,135,45,E/D,gaA/gaC,2/4,-,-1,"Pfam:PF15211,Phobius:NON_CYTOPLASMIC_DOMAIN,PANTHER:PTHR37351,AFDB-ENSP_mappings:AF-Q6UXB2-F1",-,13.65,1.541550e-02,1.541550e-02,7,0.1,donor_gain,-,-,-,-,0,"Nerve - Tibial,Brain - Spinal cord (cervical c-1)",Nerve - Tibial,0.09,Stomach - Mixed Cell,160.3,Stomach - Mixed Cell:160.3;Stomach:143.5;Stomach - Mucosa:141.8;Minor Salivary Gland:131.9;Esophagus - Mucosa:83.81,NO,0.0005614,ok,22,35,0,5,62,1,selected,-,-,top_k,-,consequence=missense_variant(+15); splice_lof=SpliceAI:0.1(+3); CADD=13.65(+2); EAS_AF=1.541550e-02; frequency(-23); domain(+3); total=+0,19:42433801:T:G,29.743000000000002,1,,0.4377713968528168,49.42224634204917,2 chr1,15660052,T,C,-,1,0.5,2,-1.164,40,-,0,10.184,-,1,0.5,60,0,17.73,-,-0.1193,0.97,1,POLYMORPHIC,13,13,HIGH,0.666667,13,26,EH38E2788758,pELS,1,ENCODE_SCREEN_v4_GRCh38,-,-,-,-,-,No,-,-,"13,26",39,99,1|0,-,-,-,"699,0,296",-,-,-,DDI2,"DDI2,RSC1A1",ENST00000480945.6,NM_032341.5,protein_coding,YES,MANE_Select,NM_032341.5,-,P1,2,CCDS30607.1,1,-,3_prime_UTR_variant,MODIFIER,c.*262T>C,-,1675,-,-,-,-,10/10,-,1,-,-,16.17,0.4303,0.05091,-,0,-,-,-,-,-,0,"Nerve - Tibial,Brain - Spinal cord (cervical c-1)",Nerve - Tibial,1.78,Cells - EBV-transformed lymphocytes,9.25,Cells - EBV-transformed lymphocytes:9.25;Esophagus - Mucosa:7.86;Cells - Cultured fibroblasts:6.465;Minor Salivary Gland:5.62;Skin - Sun Exposed (Lower leg):4.61,NO,0.1924,ok,0,35,10,3,48,1,selected,-,-,top_k,-,consequence=3_prime_utr_variant(-3); CADD=16.17(+2); EAS_AF=0.05091; frequency(-35); total=-36,1:15660052:T:C,21.142000000000003,4,,0.4574024662213867,35.44065934189303,4 diff --git a/modules/pixi_report/report/wide_table.py b/modules/pixi_report/report/wide_table.py index db9451c..e198e2f 100644 --- a/modules/pixi_report/report/wide_table.py +++ b/modules/pixi_report/report/wide_table.py @@ -96,7 +96,7 @@ def _row_to_variant(row: dict[str, str], *, use_rank_1: bool) -> VariantRecord: pathogenic_rank=_resolve_pathogenic_rank(row, use_rank_1=use_rank_1), ppi_final=_display(row.get("ppi_final")), evidence_summary=_display(row.get("evidence_summary")), - genos_evee=_display(row.get("GENOS-EVEE")), + genos_evee=_display(row.get("GENOS-VarRisk")), ) diff --git a/modules/pixi_report/scripts/export_report.py b/modules/pixi_report/scripts/export_report.py index 87b002d..b991066 100644 --- a/modules/pixi_report/scripts/export_report.py +++ b/modules/pixi_report/scripts/export_report.py @@ -23,7 +23,7 @@ CJK_MAIN_FONT = "Noto Sans SC" CJK_MONO_FONT = "Noto Sans SC" -_GENOS_MUTATION_LABELS = frozenset({"Genos-Mutation", "GENOS-EVEE"}) +_GENOS_MUTATION_LABELS = frozenset({"Genos-Mutation", "GENOS-VarRisk"}) _VARIANT_LIST_HEADER_COMPACT = ( "| 变异 | 转录本 / 后果 | Genos-Mutation | 评分 | ClinVar(VAF) |\n" From 946dff56c2e611b840ac5fa6afe0736e4ff2c6e8 Mon Sep 17 00:00:00 2001 From: lzr098 Date: Sat, 27 Jun 2026 22:12:18 +0800 Subject: [PATCH 2/8] Potential fix for pull request finding Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- modules/pixi_report/examples/demo_case/output/report.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/modules/pixi_report/examples/demo_case/output/report.md b/modules/pixi_report/examples/demo_case/output/report.md index e556aea..06f02d5 100644 --- a/modules/pixi_report/examples/demo_case/output/report.md +++ b/modules/pixi_report/examples/demo_case/output/report.md @@ -484,7 +484,7 @@ evidence_score: -36 |------|------| | 转录本 | ENST00000372827.8 | | RefSeq | NM_001720.5 | -| GENOS-VarRisk | - | +| Genos-Mutation | - | | HGVSc | c.673+4491A>G | | HGVSp | - | | VEP 后果 | intron_variant | From f91b1d53599d0dfaec31808b77084ba8a2d33791 Mon Sep 17 00:00:00 2001 From: lzr098 Date: Sat, 27 Jun 2026 22:12:28 +0800 Subject: [PATCH 3/8] Potential fix for pull request finding Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- modules/pixi_report/examples/demo_case/output/report.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/modules/pixi_report/examples/demo_case/output/report.md b/modules/pixi_report/examples/demo_case/output/report.md index 06f02d5..b4b9d6f 100644 --- a/modules/pixi_report/examples/demo_case/output/report.md +++ b/modules/pixi_report/examples/demo_case/output/report.md @@ -598,7 +598,7 @@ evidence_score: -38 |------|------| | 转录本 | ENST00000390673.2 | | RefSeq | NM_176888.2 | -| GENOS-VarRisk | - | +| Genos-Mutation | - | | HGVSc | c.889A>G | | HGVSp | p.Met297Val | | VEP 后果 | missense_variant | From e266fb77f6385dbd2f03bd3cac583467be3014ae Mon Sep 17 00:00:00 2001 From: OpenRare2026 Date: Mon, 29 Jun 2026 16:19:36 +0800 Subject: [PATCH 4/8] Potential fix for pull request finding Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- modules/pixi_report/scripts/export_report.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/modules/pixi_report/scripts/export_report.py b/modules/pixi_report/scripts/export_report.py index b991066..e5fd64a 100644 --- a/modules/pixi_report/scripts/export_report.py +++ b/modules/pixi_report/scripts/export_report.py @@ -23,7 +23,7 @@ CJK_MAIN_FONT = "Noto Sans SC" CJK_MONO_FONT = "Noto Sans SC" -_GENOS_MUTATION_LABELS = frozenset({"Genos-Mutation", "GENOS-VarRisk"}) +_GENOS_MUTATION_LABELS = frozenset({"Genos-Mutation", "GENOS-VarRisk", "GENOS-EVEE"}) _VARIANT_LIST_HEADER_COMPACT = ( "| 变异 | 转录本 / 后果 | Genos-Mutation | 评分 | ClinVar(VAF) |\n" From f99a5397154dd617ca95f4f1de205d2870ad5793 Mon Sep 17 00:00:00 2001 From: OpenRare2026 Date: Mon, 29 Jun 2026 16:20:36 +0800 Subject: [PATCH 5/8] Potential fix for pull request finding Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- modules/pipeline/scripts/run_full_pipeline.sh | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/modules/pipeline/scripts/run_full_pipeline.sh b/modules/pipeline/scripts/run_full_pipeline.sh index b20f3a3..4028035 100755 --- a/modules/pipeline/scripts/run_full_pipeline.sh +++ b/modules/pipeline/scripts/run_full_pipeline.sh @@ -107,7 +107,7 @@ while [[ $# -gt 0 ]]; do --pseudogene-annotation) PSEUDOGENE_ANNOTATION="${2:?}"; shift 2 ;; --hla-filter) HLA_FILTER="${2:?}"; shift 2 ;; --clinical-tissue) CLINICAL_TISSUE="${2:?}"; shift 2 ;; - --GENOS-VarRisk-db) GENOS_EVEE_DB="${2:?}"; shift 2 ;; + --genos-evee-db|--GENOS-VarRisk-db) GENOS_EVEE_DB="${2:?}"; shift 2 ;; --keep-raw-vep) KEEP_RAW_VEP="${2:?}"; shift 2 ;; --input-assembly) INPUT_ASSEMBLY="${2:?}"; shift 2 ;; --dry-run) DRY_RUN=yes; shift ;; From 7dc6b0eacdc33f410a8b7e4fcefe22a0372b67e6 Mon Sep 17 00:00:00 2001 From: OpenRare2026 Date: Mon, 29 Jun 2026 16:21:05 +0800 Subject: [PATCH 6/8] Potential fix for pull request finding Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- modules/pixi_report/examples/demo_case/output/report.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/modules/pixi_report/examples/demo_case/output/report.md b/modules/pixi_report/examples/demo_case/output/report.md index b4b9d6f..de05486 100644 --- a/modules/pixi_report/examples/demo_case/output/report.md +++ b/modules/pixi_report/examples/demo_case/output/report.md @@ -362,7 +362,7 @@ evidence_score: +0 |------|------| | 转录本 | ENST00000480945.6 | | RefSeq | NM_032341.5 | -| GENOS-VarRisk | - | +| Genos-Mutation | - | | HGVSc | c.*262T>C | | HGVSp | - | | VEP 后果 | 3_prime_UTR_variant | From af99cb5e37d6df6f735952ba499345e63066d7a9 Mon Sep 17 00:00:00 2001 From: wzl Date: Tue, 30 Jun 2026 00:23:31 +0800 Subject: [PATCH 7/8] Use GENOS-VarRisk in reports --- modules/pipeline/scripts/run_full_pipeline.sh | 2 +- modules/pixi_report/README.md | 2 +- .../pixi_report/examples/demo_case/output/report.md | 6 +++--- modules/pixi_report/scripts/export_report.py | 10 +++++----- 4 files changed, 10 insertions(+), 10 deletions(-) diff --git a/modules/pipeline/scripts/run_full_pipeline.sh b/modules/pipeline/scripts/run_full_pipeline.sh index 4028035..b20f3a3 100755 --- a/modules/pipeline/scripts/run_full_pipeline.sh +++ b/modules/pipeline/scripts/run_full_pipeline.sh @@ -107,7 +107,7 @@ while [[ $# -gt 0 ]]; do --pseudogene-annotation) PSEUDOGENE_ANNOTATION="${2:?}"; shift 2 ;; --hla-filter) HLA_FILTER="${2:?}"; shift 2 ;; --clinical-tissue) CLINICAL_TISSUE="${2:?}"; shift 2 ;; - --genos-evee-db|--GENOS-VarRisk-db) GENOS_EVEE_DB="${2:?}"; shift 2 ;; + --GENOS-VarRisk-db) GENOS_EVEE_DB="${2:?}"; shift 2 ;; --keep-raw-vep) KEEP_RAW_VEP="${2:?}"; shift 2 ;; --input-assembly) INPUT_ASSEMBLY="${2:?}"; shift 2 ;; --dry-run) DRY_RUN=yes; shift ;; diff --git a/modules/pixi_report/README.md b/modules/pixi_report/README.md index 84a27b1..34433a4 100644 --- a/modules/pixi_report/README.md +++ b/modules/pixi_report/README.md @@ -58,7 +58,7 @@ FastAPI(`POST /report/stream`)需要三个输入文件。路径可为绝对 | `clinical_best_tissue` / `gtex_transcript_top5_tissues` | 表达组织 | | `evidence_summary` | 证据摘要 | | `ppi_final` | 变异级 PPI 得分(Gene Card 展示回退来源) | -| `GENOS-VarRisk` | Genos-Mutation 评分(可选;宽表列名仍为 `GENOS-VarRisk`,报告中展示为 Genos-Mutation) | +| `GENOS-VarRisk` | GENOS-VarRisk 评分(可选;宽表列名和报告展示均为 `GENOS-VarRisk`) | 完整列表示例见 `examples/demo_case/wide_table.csv` 或 `fixtures/wide_table.csv`。 diff --git a/modules/pixi_report/examples/demo_case/output/report.md b/modules/pixi_report/examples/demo_case/output/report.md index de05486..e556aea 100644 --- a/modules/pixi_report/examples/demo_case/output/report.md +++ b/modules/pixi_report/examples/demo_case/output/report.md @@ -362,7 +362,7 @@ evidence_score: +0 |------|------| | 转录本 | ENST00000480945.6 | | RefSeq | NM_032341.5 | -| Genos-Mutation | - | +| GENOS-VarRisk | - | | HGVSc | c.*262T>C | | HGVSp | - | | VEP 后果 | 3_prime_UTR_variant | @@ -484,7 +484,7 @@ evidence_score: -36 |------|------| | 转录本 | ENST00000372827.8 | | RefSeq | NM_001720.5 | -| Genos-Mutation | - | +| GENOS-VarRisk | - | | HGVSc | c.673+4491A>G | | HGVSp | - | | VEP 后果 | intron_variant | @@ -598,7 +598,7 @@ evidence_score: -38 |------|------| | 转录本 | ENST00000390673.2 | | RefSeq | NM_176888.2 | -| Genos-Mutation | - | +| GENOS-VarRisk | - | | HGVSc | c.889A>G | | HGVSp | p.Met297Val | | VEP 后果 | missense_variant | diff --git a/modules/pixi_report/scripts/export_report.py b/modules/pixi_report/scripts/export_report.py index e5fd64a..d2d4751 100644 --- a/modules/pixi_report/scripts/export_report.py +++ b/modules/pixi_report/scripts/export_report.py @@ -23,10 +23,10 @@ CJK_MAIN_FONT = "Noto Sans SC" CJK_MONO_FONT = "Noto Sans SC" -_GENOS_MUTATION_LABELS = frozenset({"Genos-Mutation", "GENOS-VarRisk", "GENOS-EVEE"}) +_GENOS_VARRISK_LABELS = frozenset({"GENOS-VarRisk"}) _VARIANT_LIST_HEADER_COMPACT = ( - "| 变异 | 转录本 / 后果 | Genos-Mutation | 评分 | ClinVar(VAF) |\n" + "| 变异 | 转录本 / 后果 | GENOS-VarRisk | 评分 | ClinVar(VAF) |\n" "|------|---------------|----------------|------|----------------|" ) @@ -41,7 +41,7 @@ def _is_variant_list_header(cells: list[str]) -> bool: return cells[1] in ("转录本", "转录本 / 后果") and cells[2] in ( "后果", "评分", - *_GENOS_MUTATION_LABELS, + *_GENOS_VARRISK_LABELS, ) @@ -64,7 +64,7 @@ def _compact_variant_row(cells: list[str]) -> list[str] | None: elif len(cells) == 6: label, transcript, consequence, cadd, clinvar, vaf = cells[:6] genos = "-" - elif len(cells) == 5 and cells[2] not in _GENOS_MUTATION_LABELS: + elif len(cells) == 5 and cells[2] not in _GENOS_VARRISK_LABELS: label, transcript, consequence, cadd, clinvar, vaf = ( cells[0], cells[1], @@ -74,7 +74,7 @@ def _compact_variant_row(cells: list[str]) -> list[str] | None: "", ) genos = "-" - elif len(cells) == 5 and cells[2] in _GENOS_MUTATION_LABELS: + elif len(cells) == 5 and cells[2] in _GENOS_VARRISK_LABELS: return None elif len(cells) == 4 and (" · " in cells[1] or "
" in cells[1]): return None From c1f4f2a3cea74263951b76aed9a45977e8fe53b7 Mon Sep 17 00:00:00 2001 From: wzl Date: Tue, 30 Jun 2026 01:11:44 +0800 Subject: [PATCH 8/8] Fix VEP CPRA lookup for indels and rsIDs --- .../vep_runner/scripts/run_vep_to_csv.py | 97 ++++++++++++++++++- 1 file changed, 92 insertions(+), 5 deletions(-) diff --git a/modules/pipeline/modules/vep_runner/scripts/run_vep_to_csv.py b/modules/pipeline/modules/vep_runner/scripts/run_vep_to_csv.py index 6af9ebf..d65286f 100755 --- a/modules/pipeline/modules/vep_runner/scripts/run_vep_to_csv.py +++ b/modules/pipeline/modules/vep_runner/scripts/run_vep_to_csv.py @@ -805,6 +805,50 @@ def parse_uploaded_variation(value: str) -> dict[str, str]: return parsed +def unique_ordered(values: list[str]) -> list[str]: + seen: set[str] = set() + result: list[str] = [] + for value in values: + if not value or value in seen: + continue + seen.add(value) + result.append(value) + return result + + +def split_variant_ids(variant_id: str) -> list[str]: + variant_id = (variant_id or "").strip() + if not variant_id or variant_id in {".", "-"}: + return [] + return unique_ordered([variant_id, *[item.strip() for item in variant_id.split(";")]]) + + +def parse_vep_location(location: str) -> tuple[str, str] | None: + location = (location or "").strip() + if not location or ":" not in location: + return None + chrom, _, interval = location.partition(":") + start = interval.split("-", 1)[0].strip() + if not chrom or not start: + return None + return chrom, start + + +def location_lookup_keys(chrom: str, pos: str, ref: str, alt: str) -> list[str]: + keys = [f"loc:{chrom}:{pos}:{alt}"] + try: + shifted_pos = str(int(pos) + 1) + except ValueError: + return keys + + keys.append(f"loc:{chrom}:{shifted_pos}:{alt}") + if len(ref) < len(alt) and alt.startswith(ref): + keys.append(f"loc:{chrom}:{shifted_pos}:{alt[len(ref):] or '-'}") + elif len(ref) > len(alt) and ref.startswith(alt): + keys.append(f"loc:{chrom}:{shifted_pos}:-") + return keys + + def vep_uploaded_variation_keys(chrom: str, pos: str, ref: str, alt: str) -> list[str]: keys = [f"{chrom}_{pos}_{ref}/{alt}"] if len(ref) == len(alt): @@ -838,9 +882,52 @@ def original_variant_lookup_keys( back to the original VCF allele. """ keys = vep_uploaded_variation_keys(chrom, pos, ref, alt) - if variant_id and variant_id not in {".", "-"}: - keys.append(variant_id) - return keys + keys.extend(location_lookup_keys(chrom, pos, ref, alt)) + for item in split_variant_ids(variant_id): + keys.append(f"id:{item}") + keys.append(f"id:{item}|allele:{alt}") + return unique_ordered(keys) + + +def vep_row_candidate_keys(row: dict[str, str]) -> list[str]: + uploaded = row.get("Uploaded_variation", "") + allele = row.get("Allele", "") + keys: list[str] = [] + if uploaded: + keys.append(uploaded) + + parsed = parse_uploaded_variation(uploaded) + chrom = parsed.get("chrom", "") + pos = parsed.get("pos", "") + ref = parsed.get("ref", "") + alt = parsed.get("alt", "") + if chrom and pos and ref and alt: + keys.extend(vep_uploaded_variation_keys(chrom, pos, ref, alt)) + if allele and allele != alt: + keys.extend(vep_uploaded_variation_keys(chrom, pos, ref, allele)) + + location = parse_vep_location(row.get("Location", "")) + if location and allele: + loc_chrom, loc_pos = location + keys.append(f"loc:{loc_chrom}:{loc_pos}:{allele}") + + for item in split_variant_ids(uploaded): + if allele: + keys.append(f"id:{item}|allele:{allele}") + if alt: + keys.append(f"id:{item}|allele:{alt}") + keys.append(f"id:{item}") + keys.append(item) + + return unique_ordered(keys) + + +def lookup_original_variant(row: dict[str, str], original_variants) -> dict[str, str] | None: + for key in vep_row_candidate_keys(row): + value = original_variants.get(key) + if value is not None: + return value + return None def vcf_info_column_name(info_id: str) -> str: @@ -1915,7 +2002,7 @@ def convert_vep_table_to_csv( row = dict(zip(header, parts)) extra = parse_extra(row.pop("Extra", "")) row.update(extra) - original_variant = original_variants.get(row.get("Uploaded_variation", "")) + original_variant = lookup_original_variant(row, original_variants) add_requested_columns(row, extra, original_variant) extra_keys.update(extra) variant_id = row.get("Uploaded_variation", "") @@ -2026,7 +2113,7 @@ def parse_vep_table_groups( extra = parse_extra(row.pop("Extra", "")) row.update(extra) variant_id = row.get("Uploaded_variation", "") - original_variant = original_variants.get(variant_id) + original_variant = lookup_original_variant(row, original_variants) add_requested_columns(row, extra, original_variant) if current_key is not None and variant_id != current_key: yield current_key, current_rows