transfer from old repo

This commit is contained in:
Andreas Gammelgaard Damsbo 2026-08-19 09:27:27 +02:00
commit 277e2b8cf3
No known key found for this signature in database
1111 changed files with 83736 additions and 0 deletions

285
2 Longterm/260315/_targets.R Executable file
View file

@ -0,0 +1,285 @@
# Created by use_targets().
# Follow the comments below to fill in this target script.
# Then follow the manual to check and run the pipeline:
# https://books.ropensci.org/targets/walkthrough.html#inspect-the-pipeline
# Load packages required to define the pipeline:
library(targets)
library(tarchetypes) # Load other packages as needed.
# Set target options:
tar_option_set(
seed = 142,
packages = c("tibble"), # packages that your targets need to run
format = "qs" # , # Optionally set the default storage format. qs is fast.
#
# For distributed computing in tar_make(), supply a {crew} controller
# as discussed at https://books.ropensci.org/targets/crew.html.
# Choose a controller that suits your needs. For example, the following
# sets a controller with 2 workers which will run as local R processes:
#
# controller = crew::crew_controller_local(workers = 2)
#
# Alternatively, if you want workers to run on a high-performance computing
# cluster, select a controller from the {crew.cluster} package. The following
# example is a controller for Sun Grid Engine (SGE).
#
# controller = crew.cluster::crew_controller_sge(
# workers = 50,
# # Many clusters install R as an environment module, and you can load it
# # with the script_lines argument. To select a specific verison of R,
# # you may need to include a version string, e.g. "module load R/4.3.0".
# # Check with your system administrator if you are unsure.
# script_lines = "module load R"
# )
#
# Set other options as needed.
)
# tar_make_clustermq() is an older (pre-{crew}) way to do distributed computing
# in {targets}, and its configuration for your machine is below.
options(clustermq.scheduler = "multiprocess")
# tar_make_future() is an older (pre-{crew}) way to do distributed computing
# in {targets}, and its configuration for your machine is below.
future::plan(future.callr::callr)
# Run the R scripts in the R/ folder with your custom functions:
tar_source()
source(here::here("R/glmnet-reg.R")) # Source other scripts as needed.
# Replace the target list below with your own:
list(
tar_target(
name = pop_df,
command = get_clinical()
),
tar_target(
name = reg_list,
command = get_reg_ls()
),
tar_target(
name = df_deaths,
command = get_deaths(reg_list)
),
tar_target(
name = df_events,
command = get_events(reg_list)
),
tar_target(
name = df_events_deaths,
command = merge_events(list(events = df_events, deaths = df_deaths, clinical = pop_df))
),
tar_target(
name = df_treated,
command = get_treated(reg_list)
),
tar_target(
name = df_dst,
command = get_dst(reg_list, pop_df)
),
tar_target(
name = list_filtered,
command = list(all_events = df_events_deaths, clinical = pop_df, dst = df_dst)
),
tar_target(
name = df_all_data,
command = collectall(list_filtered)
),
tar_target(
name = df_all_data_formatted,
command = data_formatting(df_all_data)
),
tar_target(
name = ls_all_events,
command = all_events(list(events = df_events, deaths = df_deaths, clinical = pop_df))
),
tar_target(
name = df_event_data,
command = events_ready(df_all_data_formatted)
),
tar_target(
name = df_talos_data,
command = talos_ready(df_all_data_formatted)
),
tar_target(
name = df_talos_data_imp,
command = talos_imp(df_talos_data)
),
tar_target(
name = tbl_events_summary,
command = events_tblone(df_event_data)
),
tar_target(
name = tbl_events_cox_regression,
command = show_table_regression(df_event_data, use.mice = FALSE)
),
tar_target(
name = tbl_events_cox_regression_uv,
command = uv_cox_table(df_event_data)
),
tar_target(
name = df_event_data_small,
command = events_ready_small(df_all_data_formatted)
),
tar_target(
name = tbl_events_summary_small,
command = events_tblone(df_event_data_small)
),
tar_target(
name = tbl_events_cox_regression_small,
command = show_table_regression(df_event_data_small, use.mice = FALSE)
),
tar_target(
name = df_events_mids,
command = events_dataset(df_event_data, impute = TRUE)
),
tar_target(
name = df_events_complete,
command = complete_preds_data(df_event_data)
),
tar_target(
name = tbl_events_mids_cox_regression,
command = show_table_regression(df_event_data, use.mice = TRUE)
),
tar_target(
name = plot_events_survival_smooth,
command = df_event_data |> events_dataset(impute = FALSE) |> cox_regression(include_formula=TRUE) |> plot_survival_smooth()
)#,
# tar_target(
# name = plot_events_survival_smooth_mids,
# command = df_events_mids |> cox_regression() |> plot_survival_smooth()
# ),
# tar_target(
# name = df_pred_data,
# command = prediction_ready(df_all_data_formatted)
# ),
# tar_target(
# name = tbl_pred_summary,
# command = preds_tblone(df_pred_data)
# ),
# tar_target(
# name = tbl_pred_summary_true,
# command = df_pred_data |>
# dplyr::select(pase_0, pase_4, soc_status_nowork, fam_indk_hl, edu_level_hl) |>
# true_pred_sum_plot()
# ),
# tar_target(
# name = tbl_pred_summary_exp,
# command = df_pred_data |>
# dplyr::select(pase_0, pase_4, soc_status_nowork, fam_indk_hl, edu_level_hl) |>
# preds_tblone()
# ),
# tar_target(
# name = tbl_pred_summary_exp_sex,
# command = df_pred_data |>
# dplyr::select(reg_female, soc_status_nowork, fam_indk_hl, edu_level_hl) |>
# summary_tblone()
# ),
# tar_target(
# name = tbl_pred_summary_exp_pase,
# command = df_pred_data |> sum_pase_tables()
# ),
# tar_target(
# name = tbl_pred_summary_exp_missing_edu,
# command = df_pred_data |>
# dplyr::select(age,reg_female, reg_trombolyse, reg_trombektomi, reg_hyperten, reg_diabetes, nihss_0, pase_0, pase_4, soc_status_nowork, fam_indk_hl, edu_level_hl) |>
# who_is_missing(var="edu_level_hl")
# ),
# tar_target(
# name = tbl_pred_summary_exp_missing_bmi,
# command = df_pred_data |>
# dplyr::select(age,reg_female, reg_trombolyse, reg_trombektomi, reg_hyperten, reg_diabetes, nihss_0, pase_0, pase_4, reg_bmi, soc_status_nowork, fam_indk_hl, edu_level_hl) |>
# who_is_missing(var="reg_bmi")
# ),
# tar_target(
# name = ls_pred_models,
# command = pred_models(df_pred_data,auto.l = TRUE,weighted = FALSE)
# ),
# tar_target(
# name = ls_pred_models_anyupdown,
# command = pred_models(df_pred_data,auto.l = TRUE,weighted = FALSE,split.type="anyupdown")
# ),
# tar_target(
# name = ls_pred_models_relupdown_20,
# command = pred_models(df_pred_data,auto.l = TRUE,weighted = FALSE,split.type="relupdown",rel.bin=20)
# ),
# tar_target(
# name = ls_pred_models_relupdown_50,
# command = pred_models(df_pred_data,auto.l = TRUE,weighted = FALSE,split.type="relupdown",rel.bin=50)
# ),
# tar_target(
# name = df_pred_mids,
# command = fun_impute(data = df_pred_data, outcome.vars = c("pase_0", "pase_4"))
# ),
# tar_target(
# name = ls_pred_mids_reg,
# command = mids_regularisation(df_pred_mids)
# ),
# tar_target(
# name = ls_pred_mids_reg_anyupdown,
# command = mids_regularisation(df_pred_mids,split.type="anyupdown")
# ),
# tar_target(
# name = ls_pred_mids_reg_relupdown_20,
# command = mids_regularisation(df_pred_mids,split.type="relupdown",rel.bin=20)
# ),
# tar_target(
# name = ls_pred_mids_reg_relupdown_50,
# command = mids_regularisation(df_pred_mids,split.type="relupdown",rel.bin=50)
# ),
# tar_target(
# name = ls_pred_summary,
# command = multi_summary(ls_pred_models)
# ),
# tar_target(
# name = ls_pred_mids_summary,
# command = multi_summary(ls_pred_mids_reg)
# ),
# tar_target(
# name = tbl_preds_log_reg,
# command = pred_log_reg(df_pred_data)
# ),
# tar_target(
# name = tbl_preds_lin_reg,
# command = pred_lin_reg(df_pred_data)
# ),
# tar_target(
# name = tbl_preds_lin_imp_reg,
# command = pred_lin_reg(df_pred_mids)
# ),
# tar_target(
# name = list_pred_clusters,
# command = get_clusters(df_events_complete,rm.out = TRUE)
# ),
# tar_target(
# name = list_pred_clusters_seq,
# command = get_clusters(df_events_complete,n.cl=4)
# )#,
# tar_target(
# name = list_df_multi_grouping,
# command = multi_grouping_df_list(df_all_data_formatted,args.list=df_mega_list())
# ),
# tar_target(
# name = list_multi_group_results,
# command = multi_results_list(list_df_multi_grouping)
# ),
# tar_target(
# name = list_multi_group_results_direct,
# command = df_all_data_formatted |> multi_grouping_df_list(args.list=df_mega_list()) |> multi_cox_performance_test()
# )
#,
# tar_quarto(
# name = report,
# path = "index.qmd"
# ),
# tar_target(
# name = pa_change_export_files,
# command = quarto::quarto_render(here::here("doc/pa_change.qmd"), output_format = "docx")
# )
)
## TODO
## - modify pipeline to only perform imputation once and have the rest use this single object.