# 产物写到 ../csv/hm3_snplist.csv —— 脚本和 CSV 是 content/tessera/data/ 下固定的兄弟目录,
# 所以按脚本自身定位,不依赖你在哪个目录敲这条命令。csv/ 不进仓库(见 .gitignore)。
#
# Rscript 时路径在 --file= 里,source() 时在 sys.frame()$ofile 里,两种都要认:
# 只取其中一种的话,另一种跑法会静默地把 CSV 写到当前目录去。
script_dir <- local({
arg <- grep("^--file=", commandArgs(trailingOnly = FALSE), value = TRUE)
path <- if (length(arg)) sub("^--file=", "", arg[[1L]]) else sys.frame(1)$ofile
dirname(normalizePath(path, mustWork = TRUE))
})
out_csv <- file.path(script_dir, "..", "csv", "hm3_snplist.csv")
dir.create(dirname(out_csv), recursive = TRUE, showWarnings = FALSE)
# Generate the HapMap3 SNP allele list dataset for Tessera.
# Rscript content/tessera/data/script/hm3_snplist.R
out_gz <- file.path(dirname(out_csv), "w_hm3.snplist.gz")
url <- "https://zenodo.org/records/7773502/files/w_hm3.snplist.gz?download=1"
evanverse::download_url(url, out_gz, overwrite = FALSE)
hm3 <- utils::read.table(
gzfile(out_gz),
header = FALSE,
sep = "\t",
col.names = c("snp", "a1", "a2"),
colClasses = "character",
quote = "",
comment.char = ""
)
if (nrow(hm3) > 0L && identical(unname(as.character(hm3[1, ])), c("SNP", "A1", "A2"))) {
hm3 <- hm3[-1L, , drop = FALSE]
}
expected_cols <- c("snp", "a1", "a2")
missing_cols <- setdiff(expected_cols, names(hm3))
if (length(missing_cols) > 0L) {
stop(
"HapMap3 SNP list is missing columns: ",
paste(missing_cols, collapse = ", ")
)
}
hm3 <- hm3[, expected_cols]
utils::write.csv(hm3, out_csv, row.names = FALSE, na = "")
message("Wrote ", out_csv, " with ", nrow(hm3), " rows and ", ncol(hm3), " columns.")