# Example: Wide to long format nesting demonstrations
# Example 1: Basic nesting by group
w2l_nest(
data = iris, # Input dataset
by = "Species" # Group by Species column
)
#> Species data
#> <fctr> <list>
#> 1: setosa <data.table[50x4]>
#> 2: versicolor <data.table[50x4]>
#> 3: virginica <data.table[50x4]>
# Example 2: Nest specific columns with numeric indices
w2l_nest(
data = iris, # Input dataset
cols = 1:4, # Select first 4 columns to nest
by = "Species" # Group by Species column
)
#> name Species data
#> <char> <fctr> <list>
#> 1: Sepal.Length setosa <data.table[50x1]>
#> 2: Sepal.Length versicolor <data.table[50x1]>
#> 3: Sepal.Length virginica <data.table[50x1]>
#> 4: Sepal.Width setosa <data.table[50x1]>
#> 5: Sepal.Width versicolor <data.table[50x1]>
#> 6: Sepal.Width virginica <data.table[50x1]>
#> 7: Petal.Length setosa <data.table[50x1]>
#> 8: Petal.Length versicolor <data.table[50x1]>
#> 9: Petal.Length virginica <data.table[50x1]>
#> 10: Petal.Width setosa <data.table[50x1]>
#> 11: Petal.Width versicolor <data.table[50x1]>
#> 12: Petal.Width virginica <data.table[50x1]>
# Example 3: Nest specific columns with column names
w2l_nest(
data = iris, # Input dataset
cols = c("Sepal.Length", # Select columns by name
"Sepal.Width",
"Petal.Length"),
by = 5 # Group by column index 5 (Species)
)
#> name Species data
#> <char> <fctr> <list>
#> 1: Sepal.Length setosa <data.table[50x2]>
#> 2: Sepal.Length versicolor <data.table[50x2]>
#> 3: Sepal.Length virginica <data.table[50x2]>
#> 4: Sepal.Width setosa <data.table[50x2]>
#> 5: Sepal.Width versicolor <data.table[50x2]>
#> 6: Sepal.Width virginica <data.table[50x2]>
#> 7: Petal.Length setosa <data.table[50x2]>
#> 8: Petal.Length versicolor <data.table[50x2]>
#> 9: Petal.Length virginica <data.table[50x2]>
# Returns similar structure to Example 2
# Example: Wide to long format splitting demonstrations
# Example 1: Basic splitting by Species
w2l_split(
data = iris, # Input dataset
by = "Species" # Split by Species column
) |>
lapply(head) # Show first 6 rows of each split
#> $setosa
#> Sepal.Length Sepal.Width Petal.Length Petal.Width
#> <num> <num> <num> <num>
#> 1: 5.1 3.5 1.4 0.2
#> 2: 4.9 3.0 1.4 0.2
#> 3: 4.7 3.2 1.3 0.2
#> 4: 4.6 3.1 1.5 0.2
#> 5: 5.0 3.6 1.4 0.2
#> 6: 5.4 3.9 1.7 0.4
#>
#> $versicolor
#> Sepal.Length Sepal.Width Petal.Length Petal.Width
#> <num> <num> <num> <num>
#> 1: 7.0 3.2 4.7 1.4
#> 2: 6.4 3.2 4.5 1.5
#> 3: 6.9 3.1 4.9 1.5
#> 4: 5.5 2.3 4.0 1.3
#> 5: 6.5 2.8 4.6 1.5
#> 6: 5.7 2.8 4.5 1.3
#>
#> $virginica
#> Sepal.Length Sepal.Width Petal.Length Petal.Width
#> <num> <num> <num> <num>
#> 1: 6.3 3.3 6.0 2.5
#> 2: 5.8 2.7 5.1 1.9
#> 3: 7.1 3.0 5.9 2.1
#> 4: 6.3 2.9 5.6 1.8
#> 5: 6.5 3.0 5.8 2.2
#> 6: 7.6 3.0 6.6 2.1
# Example 2: Split specific columns using numeric indices
w2l_split(
data = iris, # Input dataset
cols = 1:3, # Select first 3 columns to split
by = 5 # Split by column index 5 (Species)
) |>
lapply(head) # Show first 6 rows of each split
#> $Sepal.Length_setosa
#> Petal.Width value
#> <num> <num>
#> 1: 0.2 5.1
#> 2: 0.2 4.9
#> 3: 0.2 4.7
#> 4: 0.2 4.6
#> 5: 0.2 5.0
#> 6: 0.4 5.4
#>
#> $Sepal.Length_versicolor
#> Petal.Width value
#> <num> <num>
#> 1: 1.4 7.0
#> 2: 1.5 6.4
#> 3: 1.5 6.9
#> 4: 1.3 5.5
#> 5: 1.5 6.5
#> 6: 1.3 5.7
#>
#> $Sepal.Length_virginica
#> Petal.Width value
#> <num> <num>
#> 1: 2.5 6.3
#> 2: 1.9 5.8
#> 3: 2.1 7.1
#> 4: 1.8 6.3
#> 5: 2.2 6.5
#> 6: 2.1 7.6
#>
#> $Sepal.Width_setosa
#> Petal.Width value
#> <num> <num>
#> 1: 0.2 3.5
#> 2: 0.2 3.0
#> 3: 0.2 3.2
#> 4: 0.2 3.1
#> 5: 0.2 3.6
#> 6: 0.4 3.9
#>
#> $Sepal.Width_versicolor
#> Petal.Width value
#> <num> <num>
#> 1: 1.4 3.2
#> 2: 1.5 3.2
#> 3: 1.5 3.1
#> 4: 1.3 2.3
#> 5: 1.5 2.8
#> 6: 1.3 2.8
#>
#> $Sepal.Width_virginica
#> Petal.Width value
#> <num> <num>
#> 1: 2.5 3.3
#> 2: 1.9 2.7
#> 3: 2.1 3.0
#> 4: 1.8 2.9
#> 5: 2.2 3.0
#> 6: 2.1 3.0
#>
#> $Petal.Length_setosa
#> Petal.Width value
#> <num> <num>
#> 1: 0.2 1.4
#> 2: 0.2 1.4
#> 3: 0.2 1.3
#> 4: 0.2 1.5
#> 5: 0.2 1.4
#> 6: 0.4 1.7
#>
#> $Petal.Length_versicolor
#> Petal.Width value
#> <num> <num>
#> 1: 1.4 4.7
#> 2: 1.5 4.5
#> 3: 1.5 4.9
#> 4: 1.3 4.0
#> 5: 1.5 4.6
#> 6: 1.3 4.5
#>
#> $Petal.Length_virginica
#> Petal.Width value
#> <num> <num>
#> 1: 2.5 6.0
#> 2: 1.9 5.1
#> 3: 2.1 5.9
#> 4: 1.8 5.6
#> 5: 2.2 5.8
#> 6: 2.1 6.6
# Example 3: Split specific columns using column names
list_res <- w2l_split(
data = iris, # Input dataset
cols = c("Sepal.Length", # Select columns by name
"Sepal.Width"),
by = "Species" # Split by Species column
)
lapply(list_res, head) # Show first 6 rows of each split
#> $Sepal.Length_setosa
#> Petal.Length Petal.Width value
#> <num> <num> <num>
#> 1: 1.4 0.2 5.1
#> 2: 1.4 0.2 4.9
#> 3: 1.3 0.2 4.7
#> 4: 1.5 0.2 4.6
#> 5: 1.4 0.2 5.0
#> 6: 1.7 0.4 5.4
#>
#> $Sepal.Length_versicolor
#> Petal.Length Petal.Width value
#> <num> <num> <num>
#> 1: 4.7 1.4 7.0
#> 2: 4.5 1.5 6.4
#> 3: 4.9 1.5 6.9
#> 4: 4.0 1.3 5.5
#> 5: 4.6 1.5 6.5
#> 6: 4.5 1.3 5.7
#>
#> $Sepal.Length_virginica
#> Petal.Length Petal.Width value
#> <num> <num> <num>
#> 1: 6.0 2.5 6.3
#> 2: 5.1 1.9 5.8
#> 3: 5.9 2.1 7.1
#> 4: 5.6 1.8 6.3
#> 5: 5.8 2.2 6.5
#> 6: 6.6 2.1 7.6
#>
#> $Sepal.Width_setosa
#> Petal.Length Petal.Width value
#> <num> <num> <num>
#> 1: 1.4 0.2 3.5
#> 2: 1.4 0.2 3.0
#> 3: 1.3 0.2 3.2
#> 4: 1.5 0.2 3.1
#> 5: 1.4 0.2 3.6
#> 6: 1.7 0.4 3.9
#>
#> $Sepal.Width_versicolor
#> Petal.Length Petal.Width value
#> <num> <num> <num>
#> 1: 4.7 1.4 3.2
#> 2: 4.5 1.5 3.2
#> 3: 4.9 1.5 3.1
#> 4: 4.0 1.3 2.3
#> 5: 4.6 1.5 2.8
#> 6: 4.5 1.3 2.8
#>
#> $Sepal.Width_virginica
#> Petal.Length Petal.Width value
#> <num> <num> <num>
#> 1: 6.0 2.5 3.3
#> 2: 5.1 1.9 2.7
#> 3: 5.9 2.1 3.0
#> 4: 5.6 1.8 2.9
#> 5: 5.8 2.2 3.0
#> 6: 6.6 2.1 3.0
# Returns similar structure to Example 2
# Example: Cross-validation for nested data.table demonstrations
# Setup test data
dt_nest <- w2l_nest(
data = iris, # Input dataset
cols = 1:2 # Nest first 2 columns
)
# Example 1: Basic 2-fold cross-validation (reproducible)
nest_cv(
data = dt_nest, # Input nested data.table
v = 2, # Number of folds (2-fold CV)
seed = 123 # Reproducible folds
)
#> name id train_idx validate_idx
#> <char> <char> <list> <list>
#> 1: Sepal.Length Fold1 1,2,3,5,6,7,...[75] 4, 8, 9,11,12,15,...[75]
#> 2: Sepal.Length Fold2 4, 8, 9,11,12,15,...[75] 1,2,3,5,6,7,...[75]
#> 3: Sepal.Width Fold1 1,2,3,4,6,9,...[75] 5, 7, 8,11,12,13,...[75]
#> 4: Sepal.Width Fold2 5, 7, 8,11,12,13,...[75] 1,2,3,4,6,9,...[75]
#> train validate
#> <list> <list>
#> 1: <data.table[75x4]> <data.table[75x4]>
#> 2: <data.table[75x4]> <data.table[75x4]>
#> 3: <data.table[75x4]> <data.table[75x4]>
#> 4: <data.table[75x4]> <data.table[75x4]>
# Example 2: Repeated 2-fold CV, keeping only the split objects
nest_cv(
data = dt_nest, # Input nested data.table
v = 2, # Number of folds (2-fold CV)
repeats = 2, # Number of repetitions
seed = 123,
materialize = FALSE # No train/validate copies (saves memory)
)
#> name id id2 train_idx
#> <char> <char> <char> <list>
#> 1: Sepal.Length Repeat1 Fold1 1,2,3,5,6,7,...[75]
#> 2: Sepal.Length Repeat1 Fold2 4, 8, 9,11,12,15,...[75]
#> 3: Sepal.Length Repeat2 Fold1 1,2,3,4,6,9,...[75]
#> 4: Sepal.Length Repeat2 Fold2 5, 7, 8,11,12,13,...[75]
#> 5: Sepal.Width Repeat1 Fold1 2, 3, 7, 8, 9,11,...[75]
#> 6: Sepal.Width Repeat1 Fold2 1, 4, 5, 6,10,14,...[75]
#> 7: Sepal.Width Repeat2 Fold1 3, 9,10,14,15,16,...[75]
#> 8: Sepal.Width Repeat2 Fold2 1,2,4,5,6,7,...[75]
#> validate_idx
#> <list>
#> 1: 4, 8, 9,11,12,15,...[75]
#> 2: 1,2,3,5,6,7,...[75]
#> 3: 5, 7, 8,11,12,13,...[75]
#> 4: 1,2,3,4,6,9,...[75]
#> 5: 1, 4, 5, 6,10,14,...[75]
#> 6: 2, 3, 7, 8, 9,11,...[75]
#> 7: 1,2,4,5,6,7,...[75]
#> 8: 3, 9,10,14,15,16,...[75]
# Example 3: data.frame subsets, ready for ASReml-R / lm() / glm()
cv_df <- nest_cv(dt_nest, v = 2, seed = 123, out_type = "df")
class(cv_df$train[[1]]) # "data.frame"
#> [1] "data.frame"
# Example 4: masking-style CV (keep all rows, hide validation phenotypes)
cv_idx <- nest_cv(dt_nest, v = 2, seed = 123, materialize = FALSE)
cv_idx[dt_nest, on = "name", full := i.data] # attach the full nested table
masked <- as.data.frame(cv_idx$full[[1]]) # a copy: dt_nest stays intact
masked$value[cv_idx$validate_idx[[1]]] <- NA
sum(is.na(masked$value)) # validation records to be predicted
#> [1] 75
# Prepare example data: Convert first 3 columns of iris dataset to long format and split
dt_split <- w2l_split(data = iris, cols = 1:3)
# dt_split is now a list containing 3 data tables for Sepal.Length, Sepal.Width, and Petal.Length
# Example 1: Single cross-validation (no repeats)
split_cv(
data = dt_split, # Input list of split data
v = 3, # Set 3-fold cross-validation
repeats = 1, # Perform cross-validation once (no repeats)
seed = 123 # Reproducible folds
)
#> $Sepal.Length
#> id train_idx validate_idx
#> <char> <list> <list>
#> 1: Fold1 1, 2, 5, 7, 9,10,...[100] 3, 4, 6, 8,15,19,...[50]
#> 2: Fold2 3,4,5,6,7,8,...[100] 1, 2, 9,10,11,14,...[50]
#> 3: Fold3 1,2,3,4,6,8,...[100] 5, 7,12,13,16,17,...[50]
#> train validate
#> <list> <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#>
#> $Sepal.Width
#> id train_idx validate_idx
#> <char> <list> <list>
#> 1: Fold1 2, 4, 5, 6, 7,13,...[100] 1, 3, 8, 9,10,11,...[50]
#> 2: Fold2 1,2,3,4,6,8,...[100] 5, 7,13,14,17,21,...[50]
#> 3: Fold3 1,3,5,7,8,9,...[100] 2, 4, 6,15,18,22,...[50]
#> train validate
#> <list> <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#>
#> $Petal.Length
#> id train_idx validate_idx
#> <char> <list> <list>
#> 1: Fold1 1, 2, 8, 9,10,11,...[100] 3, 4, 5, 6, 7,12,...[50]
#> 2: Fold2 2,3,4,5,6,7,...[100] 1,10,11,17,18,19,...[50]
#> 3: Fold3 1,3,4,5,6,7,...[100] 2, 8, 9,14,15,20,...[50]
#> train validate
#> <list> <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
# Returns a list where each element contains:
# - id: fold labels (Fold1, Fold2, Fold3)
# - train_idx / validate_idx: row indices of each fold
# - train / validate: training and validation subsets
# Example 2: Repeated cross-validation
split_cv(
data = dt_split, # Input list of split data
v = 3, # Set 3-fold cross-validation
repeats = 2, # Perform cross-validation twice
seed = 123
)
#> $Sepal.Length
#> id id2 train_idx validate_idx
#> <char> <char> <list> <list>
#> 1: Repeat1 Fold1 1, 2, 5, 7, 9,10,...[100] 3, 4, 6, 8,15,19,...[50]
#> 2: Repeat1 Fold2 3,4,5,6,7,8,...[100] 1, 2, 9,10,11,14,...[50]
#> 3: Repeat1 Fold3 1,2,3,4,6,8,...[100] 5, 7,12,13,16,17,...[50]
#> 4: Repeat2 Fold1 2, 4, 5, 6, 7,13,...[100] 1, 3, 8, 9,10,11,...[50]
#> 5: Repeat2 Fold2 1,2,3,4,6,8,...[100] 5, 7,13,14,17,21,...[50]
#> 6: Repeat2 Fold3 1,3,5,7,8,9,...[100] 2, 4, 6,15,18,22,...[50]
#> train validate
#> <list> <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#> 4: <data.table[100x3]> <data.table[50x3]>
#> 5: <data.table[100x3]> <data.table[50x3]>
#> 6: <data.table[100x3]> <data.table[50x3]>
#>
#> $Sepal.Width
#> id id2 train_idx validate_idx
#> <char> <char> <list> <list>
#> 1: Repeat1 Fold1 1, 2, 8, 9,10,11,...[100] 3, 4, 5, 6, 7,12,...[50]
#> 2: Repeat1 Fold2 2,3,4,5,6,7,...[100] 1,10,11,17,18,19,...[50]
#> 3: Repeat1 Fold3 1,3,4,5,6,7,...[100] 2, 8, 9,14,15,20,...[50]
#> 4: Repeat2 Fold1 1, 4, 5, 7, 9,10,...[100] 2, 3, 6, 8,14,20,...[50]
#> 5: Repeat2 Fold2 2,3,6,7,8,9,...[100] 1, 4, 5,12,17,21,...[50]
#> 6: Repeat2 Fold3 1,2,3,4,5,6,...[100] 7, 9,10,11,13,15,...[50]
#> train validate
#> <list> <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#> 4: <data.table[100x3]> <data.table[50x3]>
#> 5: <data.table[100x3]> <data.table[50x3]>
#> 6: <data.table[100x3]> <data.table[50x3]>
#>
#> $Petal.Length
#> id id2 train_idx validate_idx
#> <char> <char> <list> <list>
#> 1: Repeat1 Fold1 1,2,4,5,6,8,...[100] 3, 7, 9,15,22,23,...[50]
#> 2: Repeat1 Fold2 1,2,3,4,5,7,...[100] 6, 8,12,14,16,17,...[50]
#> 3: Repeat1 Fold3 3, 6, 7, 8, 9,12,...[100] 1, 2, 4, 5,10,11,...[50]
#> 4: Repeat2 Fold1 1, 2, 5, 8,10,12,...[100] 3, 4, 6, 7, 9,11,...[50]
#> 5: Repeat2 Fold2 3,4,5,6,7,8,...[100] 1, 2,12,14,20,21,...[50]
#> 6: Repeat2 Fold3 1,2,3,4,6,7,...[100] 5, 8,10,15,16,25,...[50]
#> train validate
#> <list> <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#> 4: <data.table[100x3]> <data.table[50x3]>
#> 5: <data.table[100x3]> <data.table[50x3]>
#> 6: <data.table[100x3]> <data.table[50x3]>
# Returns a list where each element contains:
# - id: repeat labels (Repeat1, Repeat2)
# - id2: fold labels (Fold1, Fold2, Fold3)
# - train_idx / validate_idx, train / validate
# Example 3: Stratified CV, indices only (memory friendly)
res <- split_cv(dt_split, v = 5, strata = "Species", seed = 1,
materialize = FALSE)
# Rebuild the training set of fold 1 of the first dataset when needed
head(dt_split[[1]][res[[1]]$train_idx[[1]], ])
#> Petal.Width Species value
#> <num> <fctr> <num>
#> 1: 0.2 setosa 5.1
#> 2: 0.2 setosa 4.9
#> 3: 0.2 setosa 4.7
#> 4: 0.2 setosa 4.6
#> 5: 0.2 setosa 5.0
#> 6: 0.4 setosa 5.4
# Example data preparation: Define column names for combination
col_names <- c("Sepal.Length", "Sepal.Width", "Petal.Length")
# Example 1: Basic column-to-pairs nesting with custom separator
c2p_nest(
iris, # Input iris dataset
cols = col_names, # Columns to be combined as pairs
pairs_n = 2, # Create pairs of 2 columns
sep = "&" # Custom separator for pair names
)
#> pairs data
#> <char> <list>
#> 1: Sepal.Length&Sepal.Width <data.table[150x4]>
#> 2: Sepal.Length&Petal.Length <data.table[150x4]>
#> 3: Sepal.Width&Petal.Length <data.table[150x4]>
# Returns a nested data.table where:
# - pairs: combined column names (e.g., "Sepal.Length&Sepal.Width")
# - data: list column containing data.tables with value1, value2 columns
# Example 2: Column-to-pairs nesting with numeric indices and grouping
c2p_nest(
iris, # Input iris dataset
cols = 1:3, # First 3 columns to be combined
pairs_n = 2, # Create pairs of 2 columns
by = 5 # Group by 5th column (Species)
)
#> pairs Species data
#> <char> <fctr> <list>
#> 1: Sepal.Length-Sepal.Width setosa <data.table[50x3]>
#> 2: Sepal.Length-Sepal.Width versicolor <data.table[50x3]>
#> 3: Sepal.Length-Sepal.Width virginica <data.table[50x3]>
#> 4: Sepal.Length-Petal.Length setosa <data.table[50x3]>
#> 5: Sepal.Length-Petal.Length versicolor <data.table[50x3]>
#> 6: Sepal.Length-Petal.Length virginica <data.table[50x3]>
#> 7: Sepal.Width-Petal.Length setosa <data.table[50x3]>
#> 8: Sepal.Width-Petal.Length versicolor <data.table[50x3]>
#> 9: Sepal.Width-Petal.Length virginica <data.table[50x3]>
# Returns a nested data.table where:
# - pairs: combined column names
# - Species: grouping variable
# - data: list column containing data.tables grouped by Species
# Example: the same traits recorded on the same animals in two farms
set.seed(1)
growth <- data.frame(
animal = rep(sprintf("A%02d", 1:6), each = 2),
farm = rep(c("farm1", "farm2"), times = 6),
adg = round(rnorm(12, 900, 50)), # average daily gain
bf = round(rnorm(12, 11, 1.5), 1) # backfat
)
# Example 1: column names
r2p_nest(
growth,
names_from = "farm", # levels become columns: farm1, farm2
cols = c("adg", "bf"), # traits to pivot
id = "animal" # aligns records of the same animal
)
#> name data
#> <char> <list>
#> 1: adg <data.table[6x3]>
#> 2: bf <data.table[6x3]>
# Returns a nested data.table where:
# - name: trait names (adg, bf)
# - data: one row per animal with columns animal, farm1, farm2
# Example 2: numeric indices
r2p_nest(growth, names_from = 2, cols = 3:4, id = 1)
#> name data
#> <char> <list>
#> 1: adg <data.table[6x3]>
#> 2: bf <data.table[6x3]>
# Example: Basic nested data export workflow
# A dedicated sub-folder of tempdir() keeps the clean-up safe
out_dir <- file.path(tempdir(), "mintyr_export_nest")
# Step 1: Create nested data structure
dt_nest <- w2l_nest(
data = iris, # Input iris dataset
cols = 1:2, # Columns to be nested
by = "Species" # Grouping variable
)
# Step 2: Export nested data to files
files <- export_nest(
data = dt_nest, # Input nested data.table
cols = "data", # Column containing nested data
by = c("name", "Species"), # Columns to create directory structure
path = out_dir
)
#> [ export_nest ] Export complete. 6 file(s) written to: /tmp/RtmpuLygMH/mintyr_export_nest
# Returns (invisibly) the paths of the written files
# Directory structure: out_dir/<name>/<Species>/data.txt
files
#> [1] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Length/setosa/data.txt"
#> [2] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Length/versicolor/data.txt"
#> [3] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Length/virginica/data.txt"
#> [4] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Width/setosa/data.txt"
#> [5] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Width/versicolor/data.txt"
#> [6] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Width/virginica/data.txt"
# Clean up
unlink(out_dir, recursive = TRUE)
# Example: Export split data to files
out_dir <- file.path(tempdir(), "mintyr_export_list")
# Step 1: Create split data structure
dt_split <- w2l_split(
data = iris, # Input iris dataset
cols = 1:2, # Columns to be split
by = "Species" # Grouping variable
)
# Step 2: Export split data to files
files <- export_list(
data = dt_split, # Input list of data.tables
path = out_dir
)
#> [ export_list ] Export complete. 6 / 6 file(s) written to: /tmp/RtmpuLygMH/mintyr_export_list
# Returns (invisibly) a named vector of the written file paths
files
#> Sepal.Length_setosa
#> "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Length_setosa.txt"
#> Sepal.Length_versicolor
#> "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Length_versicolor.txt"
#> Sepal.Length_virginica
#> "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Length_virginica.txt"
#> Sepal.Width_setosa
#> "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Width_setosa.txt"
#> Sepal.Width_versicolor
#> "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Width_versicolor.txt"
#> Sepal.Width_virginica
#> "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Width_virginica.txt"
# Clean up
unlink(out_dir, recursive = TRUE)
# Example: CSV file import demonstrations
# Setup test files
csv_files <- mintyr_example(
mintyr_examples("csv_test") # Get example CSV files
)
# Example 1: Import and combine CSV files using data.table
import_csv(
csv_files, # Input CSV file paths
combine = TRUE, # Combine all files into one data.table
file_col = "_file", # Column name for file source
keep_ext = TRUE, # Include .csv extension in _file column
full_path = TRUE # Show complete file paths in _file column
)
#> _file col1 col2
#> <char> <int> <char>
#> 1: /home/runner/work/_temp/Library/mintyr/extdata/csv_test1.csv 4 d
#> 2: /home/runner/work/_temp/Library/mintyr/extdata/csv_test1.csv 5 f
#> 3: /home/runner/work/_temp/Library/mintyr/extdata/csv_test1.csv 6 e
#> 4: /home/runner/work/_temp/Library/mintyr/extdata/csv_test2.csv 15 o
#> 5: /home/runner/work/_temp/Library/mintyr/extdata/csv_test2.csv 16 p
#> 6: /home/runner/work/_temp/Library/mintyr/extdata/csv_test2.csv 17 q
#> col3
#> <lgcl>
#> 1: FALSE
#> 2: TRUE
#> 3: TRUE
#> 4: FALSE
#> 5: TRUE
#> 6: FALSE
# Example: Excel file import demonstrations
# Setup test files
xlsx_files <- mintyr_example(
mintyr_examples("xlsx_test") # Get example Excel files
)
# Example 1: Import and combine all sheets from all files
import_xlsx(
xlsx_files, # Input Excel file paths
combine = TRUE # Combine all sheets into one data.table
)
#> excel_name sheet_name col1 col2 col3
#> <char> <char> <num> <char> <lgcl>
#> 1: xlsx_test1 Sheet1 4 d FALSE
#> 2: xlsx_test1 Sheet1 5 f TRUE
#> 3: xlsx_test1 Sheet1 6 e TRUE
#> 4: xlsx_test1 Sheet2 1 a TRUE
#> 5: xlsx_test1 Sheet2 2 b FALSE
#> 6: xlsx_test1 Sheet2 3 c TRUE
#> 7: xlsx_test2 Sheet1 15 o FALSE
#> 8: xlsx_test2 Sheet1 16 p TRUE
#> 9: xlsx_test2 Sheet1 17 q FALSE
#> 10: xlsx_test2 a 7 g FALSE
#> 11: xlsx_test2 a 9 h TRUE
#> 12: xlsx_test2 a 8 i FALSE
#> 13: xlsx_test2 b 10 J FALSE
#> 14: xlsx_test2 b 11 K TRUE
#> 15: xlsx_test2 b 12 L FALSE
# Example 2: Import specific sheets separately
import_xlsx(
xlsx_files, # Input Excel file paths
combine = FALSE, # Keep sheets as separate data.tables
sheet = 2 # Only import the second sheet
)
#> $xlsx_test1_Sheet2
#> col1 col2 col3
#> <num> <char> <lgcl>
#> 1: 1 a TRUE
#> 2: 2 b FALSE
#> 3: 3 c TRUE
#>
#> $xlsx_test2_a
#> col1 col2 col3
#> <num> <char> <lgcl>
#> 1: 7 g FALSE
#> 2: 9 h TRUE
#> 3: 8 i FALSE
#>
#> attr(,"source_files")
#> [1] "/home/runner/work/_temp/Library/mintyr/extdata/xlsx_test1.xlsx"
#> [2] "/home/runner/work/_temp/Library/mintyr/extdata/xlsx_test2.xlsx"
# Example 3: Multi-row header with merged cells and a title row
# (row 1 = title, rows 2-3 = header, data from row 4)
mh_file <- mintyr_example("multiheader_test.xlsx")
import_xlsx(
mh_file,
skip = 1, # Skip the title row
header_rows = 2 # Combine two header rows into one name
)
#> excel_name sheet_name ID Breed Birth date Weight (kg)_Start
#> <char> <char> <char> <char> <POSc> <num>
#> 1: multiheader_test farm_A A001 Duroc 2024-01-05 28.5
#> 2: multiheader_test farm_A A002 Duroc 2024-01-07 30.1
#> 3: multiheader_test farm_A A003 Landrace 2024-01-10 27.9
#> 4: multiheader_test farm_A A004 Landrace 2024-01-12 29.4
#> 5: multiheader_test farm_B B001 Yorkshire 2024-02-01 29.0
#> 6: multiheader_test farm_B B002 Yorkshire 2024-02-03 27.6
#> 7: multiheader_test farm_B B003 Duroc 2024-02-04 31.2
#> Weight (kg)_End Backfat (mm)_P2 Backfat (mm)_Loin Note
#> <num> <num> <num> <char>
#> 1: 102.3 11.2 56.1 <NA>
#> 2: 108.7 12.5 58.3 lame
#> 3: 99.5 10.8 54.7 <NA>
#> 4: 104.2 11.9 57.0 <NA>
#> 5: 101.8 12.1 55.4 <NA>
#> 6: 98.4 10.9 53.9 <NA>
#> 7: 110.5 13.0 59.2 culled
# The last column has an empty, unmerged upper cell: it is named "Note".
# The heuristic fill would wrongly call it "Backfat (mm)_Note":
names(import_xlsx(mh_file, sheet = 1, skip = 1, header_rows = 2,
header_fill = "right"))
#> [1] "excel_name" "sheet_name" "ID"
#> [4] "Breed" "Birth date" "Weight (kg)_Start"
#> [7] "Weight (kg)_End" "Backfat (mm)_P2" "Backfat (mm)_Loin"
#> [10] "Backfat (mm)_Note"
# Example 4: Batch import that skips unreadable files instead of stopping
bad_file <- tempfile(fileext = ".xlsx")
writeLines("not a workbook", bad_file)
res <- suppressWarnings(
import_xlsx(c(mh_file, bad_file), skip = 1, header_rows = 2, on_error = "warn")
)
unique(res$excel_name)
#> [1] "multiheader_test"
unlink(bad_file)
# Example 1: A plain data.frame -> one workbook, one sheet
out_file <- file.path(tempdir(), "mtcars.xlsx")
export_xlsx(mtcars, path = out_file, sheet_name = "mtcars")
invisible(file.remove(out_file))
# Example 2: data.table input works exactly the same way
out_file <- file.path(tempdir(), "mtcars_dt.xlsx")
export_xlsx(data.table::as.data.table(mtcars),
path = out_file, sheet_name = "mtcars")
invisible(file.remove(out_file))
# Example 3: One sheet per group in a single workbook
# Each Species value becomes a sheet; keep the Species column
out_file <- file.path(tempdir(), "iris_by_species.xlsx")
export_xlsx(iris, path = out_file, file_col = "Species", drop_cols = FALSE)
invisible(file.remove(out_file))
# Example 4: One file per group (directory mode: no .xlsx extension)
out_dir <- file.path(tempdir(), "iris_by_species")
out_files <- export_xlsx(iris, path = out_dir, file_col = "Species")
basename(out_files)
#> [1] "setosa.xlsx" "versicolor.xlsx" "virginica.xlsx"
unlink(out_dir, recursive = TRUE)
# Example 5: Round-trip the layout produced by import_xlsx(combine = TRUE)
# Rows are routed back to their original file and sheet
combined <- data.frame(
excel_name = c("sales", "sales", "costs"),
sheet_name = c("2024", "2025", "2024"),
amount = c(100, 120, 80)
)
out_dir <- file.path(tempdir(), "roundtrip")
out_files <- export_xlsx(combined, path = out_dir) # sales.xlsx, costs.xlsx
basename(out_files)
#> [1] "sales.xlsx" "costs.xlsx"
unlink(out_dir, recursive = TRUE)
# Example 6: Named list -> single workbook, one sheet per element
out_file <- file.path(tempdir(), "combined.xlsx")
res <- list(res1 = iris, res2 = mtcars)
export_xlsx(res, path = out_file)
invisible(file.remove(out_file))
# Example 7: Named list -> directory, one file per element
out_dir <- file.path(tempdir(), "combined")
out_files <- export_xlsx(res, path = out_dir) # res1.xlsx, res2.xlsx
basename(out_files)
#> [1] "res1.xlsx" "res2.xlsx"
unlink(out_dir, recursive = TRUE)
# Example 1: Basic usage with single trait
# This example selects the top 10% of observations based on Petal.Width
# keep_data=TRUE returns both summary statistics and the filtered data
top_perc(iris,
perc = 0.1, # Select top 10%
cols = c("Petal.Width"), # Column to analyze
keep_data = TRUE) # Return both stats and filtered data
#> $perc_0.1
#> $perc_0.1$stat
#> variable n min max mean median sd se cv
#> 1 Petal.Width 17 2.2 2.5 2.335294 2.3 0.09963167 0.02416423 0.04266344
#> selection
#> 1 top_10%
#>
#> $perc_0.1$data
#> Sepal.Length Sepal.Width Petal.Length Species variable value
#> 1 6.3 3.3 6.0 virginica Petal.Width 2.5
#> 2 6.5 3.0 5.8 virginica Petal.Width 2.2
#> 3 7.2 3.6 6.1 virginica Petal.Width 2.5
#> 4 5.8 2.8 5.1 virginica Petal.Width 2.4
#> 5 6.4 3.2 5.3 virginica Petal.Width 2.3
#> 6 7.7 3.8 6.7 virginica Petal.Width 2.2
#> 7 7.7 2.6 6.9 virginica Petal.Width 2.3
#> 8 6.9 3.2 5.7 virginica Petal.Width 2.3
#> 9 6.4 2.8 5.6 virginica Petal.Width 2.2
#> 10 7.7 3.0 6.1 virginica Petal.Width 2.3
#> 11 6.3 3.4 5.6 virginica Petal.Width 2.4
#> 12 6.7 3.1 5.6 virginica Petal.Width 2.4
#> 13 6.9 3.1 5.1 virginica Petal.Width 2.3
#> 14 6.8 3.2 5.9 virginica Petal.Width 2.3
#> 15 6.7 3.3 5.7 virginica Petal.Width 2.5
#> 16 6.7 3.0 5.2 virginica Petal.Width 2.3
#> 17 6.2 3.4 5.4 virginica Petal.Width 2.3
# Example 2: Using grouping with 'by' parameter
# This example performs the same analysis but separately for each Species
# Returns a data.frame of summary statistics, one row per Species
top_perc(iris,
perc = 0.1, # Select top 10%
cols = c("Petal.Width"), # Column to analyze
by = "Species") # Group by Species
#> Species variable n min max mean median sd se
#> 1 setosa Petal.Width 9 0.4 0.6 0.4333333 0.40 0.07071068 0.02357023
#> 2 versicolor Petal.Width 5 1.6 1.8 1.6600000 1.60 0.08944272 0.04000000
#> 3 virginica Petal.Width 6 2.4 2.5 2.4500000 2.45 0.05477226 0.02236068
#> cv selection
#> 1 0.16317849 top_10%
#> 2 0.05388116 top_10%
#> 3 0.02235602 top_10%
# Example 1: Common statistics for every numeric column of iris
desc_stats(iris)
#> variable n min max median iqr mean sd se
#> <char> <int> <num> <num> <num> <num> <num> <num> <num>
#> 1: Sepal.Length 150 4.3 7.9 5.80 1.3 5.843333 0.8280661 0.06761132
#> 2: Sepal.Width 150 2.0 4.4 3.00 0.5 3.057333 0.4358663 0.03558833
#> 3: Petal.Length 150 1.0 6.9 4.35 3.5 3.758000 1.7652982 0.14413600
#> 4: Petal.Width 150 0.1 2.5 1.30 1.5 1.199333 0.7622377 0.06223645
#> ci
#> <num>
#> 1: 0.13360085
#> 2: 0.07032302
#> 3: 0.28481463
#> 4: 0.12298004
# Example 2: Mean and SD by group, with a total row over all records
desc_stats(
iris,
by = "Species", # Grouping column
type = "mean_sd", # Preset: n, mean, sd
total = TRUE, # n = sum of the groups, mean / sd of all records
digits = 2 # Round the statistics
)
#> Species variable n mean sd
#> <fctr> <char> <int> <num> <num>
#> 1: setosa Sepal.Length 50 5.01 0.35
#> 2: versicolor Sepal.Length 50 5.94 0.52
#> 3: virginica Sepal.Length 50 6.59 0.64
#> 4: Total Sepal.Length 150 5.84 0.83
#> 5: setosa Sepal.Width 50 3.43 0.38
#> 6: versicolor Sepal.Width 50 2.77 0.31
#> 7: virginica Sepal.Width 50 2.97 0.32
#> 8: Total Sepal.Width 150 3.06 0.44
#> 9: setosa Petal.Length 50 1.46 0.17
#> 10: versicolor Petal.Length 50 4.26 0.47
#> 11: virginica Petal.Length 50 5.55 0.55
#> 12: Total Petal.Length 150 3.76 1.77
#> 13: setosa Petal.Width 50 0.25 0.11
#> 14: versicolor Petal.Width 50 1.33 0.20
#> 15: virginica Petal.Width 50 2.03 0.27
#> 16: Total Petal.Width 150 1.20 0.76
#> 17: setosa Total 200 NA NA
#> 18: versicolor Total 200 NA NA
#> 19: virginica Total 200 NA NA
#> 20: Total Total 600 NA NA
#> Species variable n mean sd
#> <fctr> <char> <int> <num> <num>
# Example 3: Count table with row and column totals
# (counts are the only statistic that adds up across variables)
desc_stats(
iris,
by = "Species",
stats = "n",
total = TRUE, # Total row and Total column
shape = "wide"
)
#> Species Sepal.Length Sepal.Width Petal.Length Petal.Width Total
#> <fctr> <int> <int> <int> <int> <int>
#> 1: setosa 50 50 50 50 200
#> 2: versicolor 50 50 50 50 200
#> 3: virginica 50 50 50 50 200
#> 4: Total 150 150 150 150 600
# Without groups the total adds up the counts of all variables
desc_stats(iris, type = "mean_sd", total = TRUE, digits = 2)
#> variable n mean sd
#> <char> <int> <num> <num>
#> 1: Sepal.Length 150 5.84 0.83
#> 2: Sepal.Width 150 3.06 0.44
#> 3: Petal.Length 150 3.76 1.77
#> 4: Petal.Width 150 1.20 0.76
#> 5: Total 600 NA NA
# Example 3b: Report table - groups in rows, traits in columns, "mean ± sd"
desc_stats(
mtcars,
cols = c("mpg", "hp", "wt"),
by = "cyl",
fmt = "{mean} ± {sd}", # Report-ready text
total = TRUE,
shape = "wide" # One row per group, one column per trait
)
#> cyl mpg hp wt
#> <char> <char> <char> <char>
#> 1: 4 26.66 ± 4.51 82.64 ± 20.93 2.29 ± 0.57
#> 2: 6 19.74 ± 1.45 122.29 ± 24.26 3.12 ± 0.36
#> 3: 8 15.10 ± 2.56 209.21 ± 50.98 4.00 ± 0.76
#> 4: Total 20.09 ± 6.03 146.69 ± 68.56 3.22 ± 0.98
# Example 4: Several templates, per-placeholder decimals, CV in percent and
# percentiles
desc_stats(
mtcars,
cols = c("mpg", "wt"),
by = "am",
fmt = c(N = "{n}",
"Mean ± SD" = "{mean:1} ± {sd:1}",
"CV" = "{cv:1%}",
"Median [P2.5, P97.5]" = "{median:1} [{q2.5:1}, {q97.5:1}]"),
total = TRUE
)
#> am variable N Mean ± SD CV Median [P2.5, P97.5]
#> <char> <char> <char> <char> <char> <char>
#> 1: 0 mpg 19 17.1 ± 3.8 22.4% 17.3 [10.4, 23.7]
#> 2: 1 mpg 13 24.4 ± 6.2 25.3% 22.8 [15.2, 33.4]
#> 3: Total mpg 32 20.1 ± 6.0 30.0% 19.2 [10.4, 32.7]
#> 4: 0 wt 19 3.8 ± 0.8 20.6% 3.5 [2.8, 5.4]
#> 5: 1 wt 13 2.4 ± 0.6 25.6% 2.3 [1.5, 3.4]
#> 6: Total wt 32 3.2 ± 1.0 30.4% 3.3 [1.6, 5.4]
#> 7: 0 Total 38 <NA> <NA> <NA>
#> 8: 1 Total 26 <NA> <NA> <NA>
#> 9: Total Total 64 <NA> <NA> <NA>
# Example 5: Quantiles, returned as a data.frame
desc_stats(iris, cols = 1:2, type = "quantile",
probs = c(0.05, 0.5, 0.95), out_type = "df")
#> variable n q5 q50 q95
#> 1 Sepal.Length 150 4.600 5.8 7.255
#> 2 Sepal.Width 150 2.345 3.0 3.800
paths <- c("C:/Users/foo/Documents/report.xlsx",
"/home/user/.bashrc",
"relative/path/to/data.csv",
".hidden.tar.gz",
NA_character_)
# Mode B: filename only, extension stripped (default)
get_path_info(paths)
#> [1] "report" ".bashrc" "data" ".hidden.tar" NA
# Mode B: filename only, extension preserved
get_path_info(paths, rm_extension = FALSE)
#> [1] "report.xlsx" ".bashrc" "data.csv" ".hidden.tar.gz"
#> [5] NA
# Mode B: full normalised path, extension stripped
get_path_info(paths, rm_path = FALSE)
#> [1] "C:/Users/foo/Documents/report" "/home/user/.bashrc"
#> [3] "relative/path/to/data" ".hidden.tar"
#> [5] NA
# Mode A: extract the 2nd path segment
get_path_info(paths, n = 2)
#> [1] "foo" "user" "path" NA NA
# Mode A: extract the last segment with extension stripped (n = -1 linkage)
get_path_info(paths, n = -1, rm_extension = TRUE)
#> [1] "report" ".bashrc" "data" ".hidden.tar" NA
# Mode A: range extraction
get_path_info(paths, n = c(2, 3))
#> [1] "foo/Documents" "user/.bashrc" "path/to" NA
#> [5] NA
# Example: Number formatting demonstrations
# Setup test data
dt <- data.table::data.table(
a = c(0.1234, 0.5678), # Numeric column 1
b = c(0.2345, 0.6789), # Numeric column 2
c = c("text1", "text2") # Text column
)
# Example 1: Format all numeric columns
format_digits(
dt, # Input data table
digits = 2 # Round to 2 decimal places
)
#> a b c
#> <char> <char> <char>
#> 1: 0.12 0.23 text1
#> 2: 0.57 0.68 text2
# Example 2: Format specific column as percentage
format_digits(
dt, # Input data table
cols = c("a"), # Only format column 'a'
digits = 2, # Round to 2 decimal places
percentage = TRUE # Convert to percentage
)
#> a b c
#> <char> <num> <char>
#> 1: 12.34% 0.2345 text1
#> 2: 56.78% 0.6789 text2
# Get path to an example file
mintyr_example("csv_test1.csv")
#> [1] "/home/runner/work/_temp/Library/mintyr/extdata/csv_test1.csv"
# List all example files
mintyr_examples()
#> [1] "csv_test1.csv" "csv_test2.csv" "multiheader_test.xlsx"
#> [4] "xlsx_test1.xlsx" "xlsx_test2.xlsx"