Code
source(here::here("Scripts", "00_packages.R"))
source(here::here("Scripts", "01_paths.R"))
source(here::here("Scripts", "05_phylogeny.R"))Input Effect sizes and reconciled species names.
Action Count links among effect sizes, references, systems, and species.
Output Dependency tables used to define grouping terms.
Next Check the phylogenetic matrix.
source(here::here("Scripts", "00_packages.R"))
source(here::here("Scripts", "01_paths.R"))
source(here::here("Scripts", "05_phylogeny.R"))dat_es <- readRDS(here::here("Rdata", "effect_sizes", "proceed_lnm_safe.rds"))
name_map <- readRDS(here::here("Rdata", "phylogeny", "proceed_name_map.rds"))
dat_mapped <- apply_phylo_name_map(dat_es, name_map)
dat_re_diag <- dat_mapped |>
dplyr::filter(
!is.na(sp_ncbi_canonical),
nzchar(as.character(sp_ncbi_canonical))
) |>
dplyr::transmute(
es_id_db = factor(es_id_db),
ref_id = factor(ref_id),
sys_id = factor(sys_id),
sp_ncbi = factor(sp_ncbi_canonical)
) |>
tidyr::drop_na() |>
droplevels()
analysis_counts <- tibble::tibble(
Unit = c(
"Contrasts",
"Effect-size identifiers",
"References",
"Study systems",
"Canonical species"
),
N = c(
nrow(dat_re_diag),
dplyr::n_distinct(dat_re_diag$es_id_db),
dplyr::n_distinct(dat_re_diag$ref_id),
dplyr::n_distinct(dat_re_diag$sys_id),
dplyr::n_distinct(dat_re_diag$sp_ncbi)
)
)
knitr::kable(analysis_counts, caption = "Counts in the dependency dataset.")| Unit | N |
|---|---|
| Contrasts | 7186 |
| Effect-size identifiers | 7186 |
| References | 254 |
| Study systems | 1533 |
| Canonical species | 253 |
“Canonical species” means unique species labels after taxonomic reconciliation. The current data contain 261 resolved input names but 253 canonical species because eight additional subspecies or duplicate labels collapse into five shared species labels. The next chapter documents this reconciliation and distinguishes the 253 current species from the 265 tips stored in the cached phylogenetic matrix.
summarise_links <- function(data, factor_a, factor_b) {
links <- data |>
dplyr::distinct(.data[[factor_a]], .data[[factor_b]]) |>
dplyr::count(.data[[factor_a]], name = "n_b")
tibble::tibble(
`Group A` = factor_a,
`Linked group B` = factor_b,
`Levels of A` = dplyr::n_distinct(data[[factor_a]]),
`Levels of B` = dplyr::n_distinct(data[[factor_b]]),
`A levels linked to one B` = sum(links$n_b == 1),
`A levels linked to one B (%)` = round(100 * mean(links$n_b == 1), 1),
`Median B levels per A` = stats::median(links$n_b),
`Maximum B levels per A` = max(links$n_b)
)
}
group_vars <- c("es_id_db", "ref_id", "sys_id", "sp_ncbi")
dependency_table <- purrr::map_dfr(group_vars, function(a) {
purrr::map_dfr(setdiff(group_vars, a), function(b) {
summarise_links(dat_re_diag, a, b)
})
})
knitr::kable(
dependency_table,
digits = 1,
caption = "Counts and proportions for each directed pair of grouping variables."
)| Group A | Linked group B | Levels of A | Levels of B | A levels linked to one B | A levels linked to one B (%) | Median B levels per A | Maximum B levels per A |
|---|---|---|---|---|---|---|---|
| es_id_db | ref_id | 7186 | 254 | 7186 | 100.0 | 1 | 1 |
| es_id_db | sys_id | 7186 | 1533 | 7186 | 100.0 | 1 | 1 |
| es_id_db | sp_ncbi | 7186 | 253 | 7186 | 100.0 | 1 | 1 |
| ref_id | es_id_db | 254 | 7186 | 23 | 9.1 | 8 | 1284 |
| ref_id | sys_id | 254 | 1533 | 215 | 84.6 | 1 | 1161 |
| ref_id | sp_ncbi | 254 | 253 | 235 | 92.5 | 1 | 64 |
| sys_id | es_id_db | 1533 | 7186 | 1176 | 76.7 | 1 | 1156 |
| sys_id | ref_id | 1533 | 254 | 1494 | 97.5 | 1 | 10 |
| sys_id | sp_ncbi | 1533 | 253 | 1533 | 100.0 | 1 | 1 |
| sp_ncbi | es_id_db | 253 | 7186 | 68 | 26.9 | 5 | 1192 |
| sp_ncbi | ref_id | 253 | 254 | 200 | 79.1 | 1 | 13 |
| sp_ncbi | sys_id | 253 | 1533 | 198 | 78.3 | 1 | 379 |
system_table <- dplyr::bind_rows(
summarise_links(dat_re_diag, "sp_ncbi", "sys_id"),
summarise_links(dat_re_diag, "sys_id", "sp_ncbi"),
summarise_links(dat_re_diag, "ref_id", "sys_id"),
summarise_links(dat_re_diag, "sys_id", "ref_id")
)
knitr::kable(
system_table,
digits = 1,
caption = "Counts and proportions linking study systems with species and references."
)| Group A | Linked group B | Levels of A | Levels of B | A levels linked to one B | A levels linked to one B (%) | Median B levels per A | Maximum B levels per A |
|---|---|---|---|---|---|---|---|
| sp_ncbi | sys_id | 253 | 1533 | 198 | 78.3 | 1 | 379 |
| sys_id | sp_ncbi | 1533 | 253 | 1533 | 100.0 | 1 | 1 |
| ref_id | sys_id | 254 | 1533 | 215 | 84.6 | 1 | 1161 |
| sys_id | ref_id | 1533 | 254 | 1494 | 97.5 | 1 | 10 |