Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion DESCRIPTION
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
Package: mia
Type: Package
Version: 1.21.5
Version: 1.21.6
Authors@R:
c(person(given = "Tuomas", family = "Borman", role = c("aut", "cre"),
email = "tuomas.v.borman@utu.fi",
Expand Down
Binary file modified data/ibdmdb.rda
Binary file not shown.
49 changes: 46 additions & 3 deletions inst/scripts/prepare_ibdmdb.R
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
# Build compact demo objects (.rda) for IBDMDB examples/vignettes.
#
# Produces:
# data/ibdmdb.rda (ibdmdb)

Check warning on line 4 in inst/scripts/prepare_ibdmdb.R

View workflow job for this annotation

GitHub Actions / ubuntu-latest (auto)

[lintr] Remove commented code.
#
# Source raw inputs from inst/extdata and pre-process for speed/size.
#
Expand All @@ -22,9 +22,9 @@
dir.create(cache_dir, recursive = TRUE, showWarnings = FALSE)

# Stable download URLs (Zenodo)
url_meta <- "https://zenodo.org/records/18280405/files/hmp2_metadata_2018-08-20.csv"

Check warning on line 25 in inst/scripts/prepare_ibdmdb.R

View workflow job for this annotation

GitHub Actions / ubuntu-latest (auto)

[lintr] Lines should not be more than 80 characters. This line is 84 characters.
url_mtx <- "https://zenodo.org/records/18280535/files/ecs_relab.tsv"
url_mgx <- "https://zenodo.org/records/18280521/files/taxonomic_profiles_mgx.tsv"

Check warning on line 27 in inst/scripts/prepare_ibdmdb.R

View workflow job for this annotation

GitHub Actions / ubuntu-latest (auto)

[lintr] Lines should not be more than 80 characters. This line is 82 characters.

# Local cached file paths (downloaded if missing)
f_mgx <- file.path(cache_dir, "taxonomic_profiles_mgx.tsv")
Expand All @@ -44,12 +44,12 @@
# Dependencies
# ------------------------------------------------------------------------------

need <- c("data.table", "matrixStats", "SummarizedExperiment", "MultiAssayExperiment",

Check warning on line 47 in inst/scripts/prepare_ibdmdb.R

View workflow job for this annotation

GitHub Actions / ubuntu-latest (auto)

[lintr] Lines should not be more than 80 characters. This line is 86 characters.
"S4Vectors")
miss <- setdiff(need, rownames(installed.packages()))
if (length(miss)) {
stop(

Check warning on line 51 in inst/scripts/prepare_ibdmdb.R

View workflow job for this annotation

GitHub Actions / ubuntu-latest (auto)

[lintr] Indentation should be 2 spaces but is 4 spaces.
"Missing packages: ",

Check warning on line 52 in inst/scripts/prepare_ibdmdb.R

View workflow job for this annotation

GitHub Actions / ubuntu-latest (auto)

[lintr] Indentation should be 4 spaces but is 8 spaces.
paste(miss, collapse = ", "),
". Install them, then re-run this script."
)
Expand Down Expand Up @@ -153,7 +153,7 @@
))
}

md <- as.data.frame(meta_df, stringsAsFactors = FALSE, check.names = FALSE)
md <- meta_df

overlaps <- vapply(
md,
Expand Down Expand Up @@ -182,7 +182,7 @@

SummarizedExperiment::SummarizedExperiment(
assays = setNames(list(mat), assay_name),
colData = S4Vectors::DataFrame(md_sub)
colData = md_sub
)
}

Expand Down Expand Up @@ -210,6 +210,7 @@
# ------------------------------------------------------------------------------

meta_full <- read_metadata(f_meta)
meta_full <- DataFrame(meta_full, check.names = FALSE)

# ------------------------------------------------------------------------------
# Prepare 2-omic (MGX + MTX)
Expand Down Expand Up @@ -252,9 +253,51 @@
M_mgx <- cap_by_var(M_mgx, cap_mgx)
M_mtx <- cap_by_var(M_mtx, cap_mtx)

meta_df <- meta_full
meta_df <- meta_full[meta_full$data_type %in% "metagenomics", ]
se_mgx <- make_SE(M_mgx, meta_df, assay_name = "mgx")
# Split taxa into taxonomy
rd <- mia:::.parse_taxonomy(data.frame(Taxon = rownames(se_mgx)), sep = "\\|", remove.prefix = TRUE)
rownames(rd) <- rownames(se_mgx)
rowData(se_mgx) <- rd
se_mgx <- agglomerateByRanks(se_mgx)
se_mgx <- swapAltExp(se_mgx, "species", "original")

meta_df <- meta_full[meta_full$data_type %in% "metatranscriptomics", ]
se_mtx <- make_SE(M_mtx, meta_df, assay_name = "mtx")
# Create a feature metadata table
df <- data.frame(
feature = rownames(se_mtx),
stringsAsFactors = FALSE
)
df$gene_function <- sub("\\|.*$", "", df$feature)
df$taxon <- ifelse(
grepl("\\|", df$feature),
sub("^.*\\|", "", df$feature),
NA_character_
)
# Create function + bacteria identifier
df$gene_taxon <- ifelse(
!is.na(df$taxon),
paste(df$gene_function, df$taxon, sep = "|"),
NA_character_
)
df <- DataFrame(df)
rownames(df) <- rownames(se_mtx)
rowData(se_mtx) <- df
se_mtx <- as(se_mtx, "TreeSummarizedExperiment")

# Function level
altExp(se_mtx, "gene_function") <- agglomerateByVariable(
se_mtx,
by = 1,
group = "gene_function"
)
# Function + bacteria level
altExp(se_mtx, "gene_taxon") <- agglomerateByVariable(
se_mtx,
by = 1,
group = "gene_taxon"
)

mae <- MultiAssayExperiment::MultiAssayExperiment(
experiments = list(MGX = se_mgx, MTX = se_mtx)
Expand Down
Loading