Internal helpers

w2l_nest

# Example: Wide to long format nesting demonstrations

# Example 1: Basic nesting by group
w2l_nest(
  data = iris,                    # Input dataset
  by = "Species"                  # Group by Species column
)
#>       Species               data
#>        <fctr>             <list>
#> 1:     setosa <data.table[50x4]>
#> 2: versicolor <data.table[50x4]>
#> 3:  virginica <data.table[50x4]>

# Example 2: Nest specific columns with numeric indices
w2l_nest(
  data = iris,                    # Input dataset
  cols = 1:4,                   # Select first 4 columns to nest
  by = "Species"                  # Group by Species column
)
#>             name    Species               data
#>           <char>     <fctr>             <list>
#>  1: Sepal.Length     setosa <data.table[50x1]>
#>  2: Sepal.Length versicolor <data.table[50x1]>
#>  3: Sepal.Length  virginica <data.table[50x1]>
#>  4:  Sepal.Width     setosa <data.table[50x1]>
#>  5:  Sepal.Width versicolor <data.table[50x1]>
#>  6:  Sepal.Width  virginica <data.table[50x1]>
#>  7: Petal.Length     setosa <data.table[50x1]>
#>  8: Petal.Length versicolor <data.table[50x1]>
#>  9: Petal.Length  virginica <data.table[50x1]>
#> 10:  Petal.Width     setosa <data.table[50x1]>
#> 11:  Petal.Width versicolor <data.table[50x1]>
#> 12:  Petal.Width  virginica <data.table[50x1]>

# Example 3: Nest specific columns with column names
w2l_nest(
  data = iris,                    # Input dataset
  cols = c("Sepal.Length",        # Select columns by name
           "Sepal.Width",
           "Petal.Length"),
  by = 5                          # Group by column index 5 (Species)
)
#>            name    Species               data
#>          <char>     <fctr>             <list>
#> 1: Sepal.Length     setosa <data.table[50x2]>
#> 2: Sepal.Length versicolor <data.table[50x2]>
#> 3: Sepal.Length  virginica <data.table[50x2]>
#> 4:  Sepal.Width     setosa <data.table[50x2]>
#> 5:  Sepal.Width versicolor <data.table[50x2]>
#> 6:  Sepal.Width  virginica <data.table[50x2]>
#> 7: Petal.Length     setosa <data.table[50x2]>
#> 8: Petal.Length versicolor <data.table[50x2]>
#> 9: Petal.Length  virginica <data.table[50x2]>
# Returns similar structure to Example 2

w2l_split

# Example: Wide to long format splitting demonstrations

# Example 1: Basic splitting by Species
w2l_split(
  data = iris,                    # Input dataset
  by = "Species"                  # Split by Species column
) |> 
  lapply(head)                    # Show first 6 rows of each split
#> $setosa
#>    Sepal.Length Sepal.Width Petal.Length Petal.Width
#>           <num>       <num>        <num>       <num>
#> 1:          5.1         3.5          1.4         0.2
#> 2:          4.9         3.0          1.4         0.2
#> 3:          4.7         3.2          1.3         0.2
#> 4:          4.6         3.1          1.5         0.2
#> 5:          5.0         3.6          1.4         0.2
#> 6:          5.4         3.9          1.7         0.4
#> 
#> $versicolor
#>    Sepal.Length Sepal.Width Petal.Length Petal.Width
#>           <num>       <num>        <num>       <num>
#> 1:          7.0         3.2          4.7         1.4
#> 2:          6.4         3.2          4.5         1.5
#> 3:          6.9         3.1          4.9         1.5
#> 4:          5.5         2.3          4.0         1.3
#> 5:          6.5         2.8          4.6         1.5
#> 6:          5.7         2.8          4.5         1.3
#> 
#> $virginica
#>    Sepal.Length Sepal.Width Petal.Length Petal.Width
#>           <num>       <num>        <num>       <num>
#> 1:          6.3         3.3          6.0         2.5
#> 2:          5.8         2.7          5.1         1.9
#> 3:          7.1         3.0          5.9         2.1
#> 4:          6.3         2.9          5.6         1.8
#> 5:          6.5         3.0          5.8         2.2
#> 6:          7.6         3.0          6.6         2.1

# Example 2: Split specific columns using numeric indices
w2l_split(
  data = iris,                    # Input dataset
  cols = 1:3,                   # Select first 3 columns to split
  by = 5                          # Split by column index 5 (Species)
) |> 
  lapply(head)                    # Show first 6 rows of each split
#> $Sepal.Length_setosa
#>    Petal.Width value
#>          <num> <num>
#> 1:         0.2   5.1
#> 2:         0.2   4.9
#> 3:         0.2   4.7
#> 4:         0.2   4.6
#> 5:         0.2   5.0
#> 6:         0.4   5.4
#> 
#> $Sepal.Length_versicolor
#>    Petal.Width value
#>          <num> <num>
#> 1:         1.4   7.0
#> 2:         1.5   6.4
#> 3:         1.5   6.9
#> 4:         1.3   5.5
#> 5:         1.5   6.5
#> 6:         1.3   5.7
#> 
#> $Sepal.Length_virginica
#>    Petal.Width value
#>          <num> <num>
#> 1:         2.5   6.3
#> 2:         1.9   5.8
#> 3:         2.1   7.1
#> 4:         1.8   6.3
#> 5:         2.2   6.5
#> 6:         2.1   7.6
#> 
#> $Sepal.Width_setosa
#>    Petal.Width value
#>          <num> <num>
#> 1:         0.2   3.5
#> 2:         0.2   3.0
#> 3:         0.2   3.2
#> 4:         0.2   3.1
#> 5:         0.2   3.6
#> 6:         0.4   3.9
#> 
#> $Sepal.Width_versicolor
#>    Petal.Width value
#>          <num> <num>
#> 1:         1.4   3.2
#> 2:         1.5   3.2
#> 3:         1.5   3.1
#> 4:         1.3   2.3
#> 5:         1.5   2.8
#> 6:         1.3   2.8
#> 
#> $Sepal.Width_virginica
#>    Petal.Width value
#>          <num> <num>
#> 1:         2.5   3.3
#> 2:         1.9   2.7
#> 3:         2.1   3.0
#> 4:         1.8   2.9
#> 5:         2.2   3.0
#> 6:         2.1   3.0
#> 
#> $Petal.Length_setosa
#>    Petal.Width value
#>          <num> <num>
#> 1:         0.2   1.4
#> 2:         0.2   1.4
#> 3:         0.2   1.3
#> 4:         0.2   1.5
#> 5:         0.2   1.4
#> 6:         0.4   1.7
#> 
#> $Petal.Length_versicolor
#>    Petal.Width value
#>          <num> <num>
#> 1:         1.4   4.7
#> 2:         1.5   4.5
#> 3:         1.5   4.9
#> 4:         1.3   4.0
#> 5:         1.5   4.6
#> 6:         1.3   4.5
#> 
#> $Petal.Length_virginica
#>    Petal.Width value
#>          <num> <num>
#> 1:         2.5   6.0
#> 2:         1.9   5.1
#> 3:         2.1   5.9
#> 4:         1.8   5.6
#> 5:         2.2   5.8
#> 6:         2.1   6.6

# Example 3: Split specific columns using column names
list_res <- w2l_split(
  data = iris,                    # Input dataset
  cols = c("Sepal.Length",        # Select columns by name
           "Sepal.Width"),
  by = "Species"                  # Split by Species column
)
lapply(list_res, head)            # Show first 6 rows of each split
#> $Sepal.Length_setosa
#>    Petal.Length Petal.Width value
#>           <num>       <num> <num>
#> 1:          1.4         0.2   5.1
#> 2:          1.4         0.2   4.9
#> 3:          1.3         0.2   4.7
#> 4:          1.5         0.2   4.6
#> 5:          1.4         0.2   5.0
#> 6:          1.7         0.4   5.4
#> 
#> $Sepal.Length_versicolor
#>    Petal.Length Petal.Width value
#>           <num>       <num> <num>
#> 1:          4.7         1.4   7.0
#> 2:          4.5         1.5   6.4
#> 3:          4.9         1.5   6.9
#> 4:          4.0         1.3   5.5
#> 5:          4.6         1.5   6.5
#> 6:          4.5         1.3   5.7
#> 
#> $Sepal.Length_virginica
#>    Petal.Length Petal.Width value
#>           <num>       <num> <num>
#> 1:          6.0         2.5   6.3
#> 2:          5.1         1.9   5.8
#> 3:          5.9         2.1   7.1
#> 4:          5.6         1.8   6.3
#> 5:          5.8         2.2   6.5
#> 6:          6.6         2.1   7.6
#> 
#> $Sepal.Width_setosa
#>    Petal.Length Petal.Width value
#>           <num>       <num> <num>
#> 1:          1.4         0.2   3.5
#> 2:          1.4         0.2   3.0
#> 3:          1.3         0.2   3.2
#> 4:          1.5         0.2   3.1
#> 5:          1.4         0.2   3.6
#> 6:          1.7         0.4   3.9
#> 
#> $Sepal.Width_versicolor
#>    Petal.Length Petal.Width value
#>           <num>       <num> <num>
#> 1:          4.7         1.4   3.2
#> 2:          4.5         1.5   3.2
#> 3:          4.9         1.5   3.1
#> 4:          4.0         1.3   2.3
#> 5:          4.6         1.5   2.8
#> 6:          4.5         1.3   2.8
#> 
#> $Sepal.Width_virginica
#>    Petal.Length Petal.Width value
#>           <num>       <num> <num>
#> 1:          6.0         2.5   3.3
#> 2:          5.1         1.9   2.7
#> 3:          5.9         2.1   3.0
#> 4:          5.6         1.8   2.9
#> 5:          5.8         2.2   3.0
#> 6:          6.6         2.1   3.0
# Returns similar structure to Example 2

nest_cv

# Example: Cross-validation for nested data.table demonstrations

# Setup test data
dt_nest <- w2l_nest(
  data = iris,                   # Input dataset
  cols = 1:2                     # Nest first 2 columns
)

# Example 1: Basic 2-fold cross-validation (reproducible)
nest_cv(
  data = dt_nest,                # Input nested data.table
  v = 2,                         # Number of folds (2-fold CV)
  seed = 123                     # Reproducible folds
)
#>            name     id                 train_idx              validate_idx
#>          <char> <char>                    <list>                    <list>
#> 1: Sepal.Length  Fold1       1,2,3,5,6,7,...[75]  4, 8, 9,11,12,15,...[75]
#> 2: Sepal.Length  Fold2  4, 8, 9,11,12,15,...[75]       1,2,3,5,6,7,...[75]
#> 3:  Sepal.Width  Fold1       1,2,3,4,6,9,...[75]  5, 7, 8,11,12,13,...[75]
#> 4:  Sepal.Width  Fold2  5, 7, 8,11,12,13,...[75]       1,2,3,4,6,9,...[75]
#>                 train           validate
#>                <list>             <list>
#> 1: <data.table[75x4]> <data.table[75x4]>
#> 2: <data.table[75x4]> <data.table[75x4]>
#> 3: <data.table[75x4]> <data.table[75x4]>
#> 4: <data.table[75x4]> <data.table[75x4]>

# Example 2: Repeated 2-fold CV, keeping only the split objects
nest_cv(
  data = dt_nest,                # Input nested data.table
  v = 2,                         # Number of folds (2-fold CV)
  repeats = 2,                   # Number of repetitions
  seed = 123,
  materialize = FALSE            # No train/validate copies (saves memory)
)
#>            name      id    id2                 train_idx
#>          <char>  <char> <char>                    <list>
#> 1: Sepal.Length Repeat1  Fold1       1,2,3,5,6,7,...[75]
#> 2: Sepal.Length Repeat1  Fold2  4, 8, 9,11,12,15,...[75]
#> 3: Sepal.Length Repeat2  Fold1       1,2,3,4,6,9,...[75]
#> 4: Sepal.Length Repeat2  Fold2  5, 7, 8,11,12,13,...[75]
#> 5:  Sepal.Width Repeat1  Fold1  2, 3, 7, 8, 9,11,...[75]
#> 6:  Sepal.Width Repeat1  Fold2  1, 4, 5, 6,10,14,...[75]
#> 7:  Sepal.Width Repeat2  Fold1  3, 9,10,14,15,16,...[75]
#> 8:  Sepal.Width Repeat2  Fold2       1,2,4,5,6,7,...[75]
#>                 validate_idx
#>                       <list>
#> 1:  4, 8, 9,11,12,15,...[75]
#> 2:       1,2,3,5,6,7,...[75]
#> 3:  5, 7, 8,11,12,13,...[75]
#> 4:       1,2,3,4,6,9,...[75]
#> 5:  1, 4, 5, 6,10,14,...[75]
#> 6:  2, 3, 7, 8, 9,11,...[75]
#> 7:       1,2,4,5,6,7,...[75]
#> 8:  3, 9,10,14,15,16,...[75]

# Example 3: data.frame subsets, ready for ASReml-R / lm() / glm()
cv_df <- nest_cv(dt_nest, v = 2, seed = 123, out_type = "df")
class(cv_df$train[[1]])        # "data.frame"
#> [1] "data.frame"

# Example 4: masking-style CV (keep all rows, hide validation phenotypes)
cv_idx <- nest_cv(dt_nest, v = 2, seed = 123, materialize = FALSE)
cv_idx[dt_nest, on = "name", full := i.data]   # attach the full nested table
masked <- as.data.frame(cv_idx$full[[1]])      # a copy: dt_nest stays intact
masked$value[cv_idx$validate_idx[[1]]] <- NA
sum(is.na(masked$value))       # validation records to be predicted
#> [1] 75

split_cv

# Prepare example data: Convert first 3 columns of iris dataset to long format and split
dt_split <- w2l_split(data = iris, cols = 1:3)
# dt_split is now a list containing 3 data tables for Sepal.Length, Sepal.Width, and Petal.Length

# Example 1: Single cross-validation (no repeats)
split_cv(
  data = dt_split,      # Input list of split data
  v = 3,                # Set 3-fold cross-validation
  repeats = 1,          # Perform cross-validation once (no repeats)
  seed = 123            # Reproducible folds
)
#> $Sepal.Length
#>        id                  train_idx              validate_idx
#>    <char>                     <list>                    <list>
#> 1:  Fold1  1, 2, 5, 7, 9,10,...[100]  3, 4, 6, 8,15,19,...[50]
#> 2:  Fold2       3,4,5,6,7,8,...[100]  1, 2, 9,10,11,14,...[50]
#> 3:  Fold3       1,2,3,4,6,8,...[100]  5, 7,12,13,16,17,...[50]
#>                  train           validate
#>                 <list>             <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#> 
#> $Sepal.Width
#>        id                  train_idx              validate_idx
#>    <char>                     <list>                    <list>
#> 1:  Fold1  2, 4, 5, 6, 7,13,...[100]  1, 3, 8, 9,10,11,...[50]
#> 2:  Fold2       1,2,3,4,6,8,...[100]  5, 7,13,14,17,21,...[50]
#> 3:  Fold3       1,3,5,7,8,9,...[100]  2, 4, 6,15,18,22,...[50]
#>                  train           validate
#>                 <list>             <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#> 
#> $Petal.Length
#>        id                  train_idx              validate_idx
#>    <char>                     <list>                    <list>
#> 1:  Fold1  1, 2, 8, 9,10,11,...[100]  3, 4, 5, 6, 7,12,...[50]
#> 2:  Fold2       2,3,4,5,6,7,...[100]  1,10,11,17,18,19,...[50]
#> 3:  Fold3       1,3,4,5,6,7,...[100]  2, 8, 9,14,15,20,...[50]
#>                  train           validate
#>                 <list>             <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
# Returns a list where each element contains:
# - id: fold labels (Fold1, Fold2, Fold3)
# - train_idx / validate_idx: row indices of each fold
# - train / validate: training and validation subsets

# Example 2: Repeated cross-validation
split_cv(
  data = dt_split,      # Input list of split data
  v = 3,                # Set 3-fold cross-validation
  repeats = 2,          # Perform cross-validation twice
  seed = 123
)
#> $Sepal.Length
#>         id    id2                  train_idx              validate_idx
#>     <char> <char>                     <list>                    <list>
#> 1: Repeat1  Fold1  1, 2, 5, 7, 9,10,...[100]  3, 4, 6, 8,15,19,...[50]
#> 2: Repeat1  Fold2       3,4,5,6,7,8,...[100]  1, 2, 9,10,11,14,...[50]
#> 3: Repeat1  Fold3       1,2,3,4,6,8,...[100]  5, 7,12,13,16,17,...[50]
#> 4: Repeat2  Fold1  2, 4, 5, 6, 7,13,...[100]  1, 3, 8, 9,10,11,...[50]
#> 5: Repeat2  Fold2       1,2,3,4,6,8,...[100]  5, 7,13,14,17,21,...[50]
#> 6: Repeat2  Fold3       1,3,5,7,8,9,...[100]  2, 4, 6,15,18,22,...[50]
#>                  train           validate
#>                 <list>             <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#> 4: <data.table[100x3]> <data.table[50x3]>
#> 5: <data.table[100x3]> <data.table[50x3]>
#> 6: <data.table[100x3]> <data.table[50x3]>
#> 
#> $Sepal.Width
#>         id    id2                  train_idx              validate_idx
#>     <char> <char>                     <list>                    <list>
#> 1: Repeat1  Fold1  1, 2, 8, 9,10,11,...[100]  3, 4, 5, 6, 7,12,...[50]
#> 2: Repeat1  Fold2       2,3,4,5,6,7,...[100]  1,10,11,17,18,19,...[50]
#> 3: Repeat1  Fold3       1,3,4,5,6,7,...[100]  2, 8, 9,14,15,20,...[50]
#> 4: Repeat2  Fold1  1, 4, 5, 7, 9,10,...[100]  2, 3, 6, 8,14,20,...[50]
#> 5: Repeat2  Fold2       2,3,6,7,8,9,...[100]  1, 4, 5,12,17,21,...[50]
#> 6: Repeat2  Fold3       1,2,3,4,5,6,...[100]  7, 9,10,11,13,15,...[50]
#>                  train           validate
#>                 <list>             <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#> 4: <data.table[100x3]> <data.table[50x3]>
#> 5: <data.table[100x3]> <data.table[50x3]>
#> 6: <data.table[100x3]> <data.table[50x3]>
#> 
#> $Petal.Length
#>         id    id2                  train_idx              validate_idx
#>     <char> <char>                     <list>                    <list>
#> 1: Repeat1  Fold1       1,2,4,5,6,8,...[100]  3, 7, 9,15,22,23,...[50]
#> 2: Repeat1  Fold2       1,2,3,4,5,7,...[100]  6, 8,12,14,16,17,...[50]
#> 3: Repeat1  Fold3  3, 6, 7, 8, 9,12,...[100]  1, 2, 4, 5,10,11,...[50]
#> 4: Repeat2  Fold1  1, 2, 5, 8,10,12,...[100]  3, 4, 6, 7, 9,11,...[50]
#> 5: Repeat2  Fold2       3,4,5,6,7,8,...[100]  1, 2,12,14,20,21,...[50]
#> 6: Repeat2  Fold3       1,2,3,4,6,7,...[100]  5, 8,10,15,16,25,...[50]
#>                  train           validate
#>                 <list>             <list>
#> 1: <data.table[100x3]> <data.table[50x3]>
#> 2: <data.table[100x3]> <data.table[50x3]>
#> 3: <data.table[100x3]> <data.table[50x3]>
#> 4: <data.table[100x3]> <data.table[50x3]>
#> 5: <data.table[100x3]> <data.table[50x3]>
#> 6: <data.table[100x3]> <data.table[50x3]>
# Returns a list where each element contains:
# - id: repeat labels (Repeat1, Repeat2)
# - id2: fold labels (Fold1, Fold2, Fold3)
# - train_idx / validate_idx, train / validate

# Example 3: Stratified CV, indices only (memory friendly)
res <- split_cv(dt_split, v = 5, strata = "Species", seed = 1,
                materialize = FALSE)
# Rebuild the training set of fold 1 of the first dataset when needed
head(dt_split[[1]][res[[1]]$train_idx[[1]], ])
#>    Petal.Width Species value
#>          <num>  <fctr> <num>
#> 1:         0.2  setosa   5.1
#> 2:         0.2  setosa   4.9
#> 3:         0.2  setosa   4.7
#> 4:         0.2  setosa   4.6
#> 5:         0.2  setosa   5.0
#> 6:         0.4  setosa   5.4

c2p_nest

# Example data preparation: Define column names for combination
col_names <- c("Sepal.Length", "Sepal.Width", "Petal.Length")

# Example 1: Basic column-to-pairs nesting with custom separator
c2p_nest(
  iris,                   # Input iris dataset
  cols = col_names,       # Columns to be combined as pairs
  pairs_n = 2,            # Create pairs of 2 columns
  sep = "&"               # Custom separator for pair names
)
#>                        pairs                data
#>                       <char>              <list>
#> 1:  Sepal.Length&Sepal.Width <data.table[150x4]>
#> 2: Sepal.Length&Petal.Length <data.table[150x4]>
#> 3:  Sepal.Width&Petal.Length <data.table[150x4]>
# Returns a nested data.table where:
# - pairs: combined column names (e.g., "Sepal.Length&Sepal.Width")
# - data: list column containing data.tables with value1, value2 columns

# Example 2: Column-to-pairs nesting with numeric indices and grouping
c2p_nest(
  iris,                   # Input iris dataset
  cols = 1:3,             # First 3 columns to be combined
  pairs_n = 2,            # Create pairs of 2 columns
  by = 5                  # Group by 5th column (Species)
)
#>                        pairs    Species               data
#>                       <char>     <fctr>             <list>
#> 1:  Sepal.Length-Sepal.Width     setosa <data.table[50x3]>
#> 2:  Sepal.Length-Sepal.Width versicolor <data.table[50x3]>
#> 3:  Sepal.Length-Sepal.Width  virginica <data.table[50x3]>
#> 4: Sepal.Length-Petal.Length     setosa <data.table[50x3]>
#> 5: Sepal.Length-Petal.Length versicolor <data.table[50x3]>
#> 6: Sepal.Length-Petal.Length  virginica <data.table[50x3]>
#> 7:  Sepal.Width-Petal.Length     setosa <data.table[50x3]>
#> 8:  Sepal.Width-Petal.Length versicolor <data.table[50x3]>
#> 9:  Sepal.Width-Petal.Length  virginica <data.table[50x3]>
# Returns a nested data.table where:
# - pairs: combined column names
# - Species: grouping variable
# - data: list column containing data.tables grouped by Species

r2p_nest

# Example: the same traits recorded on the same animals in two farms
set.seed(1)
growth <- data.frame(
  animal = rep(sprintf("A%02d", 1:6), each = 2),
  farm   = rep(c("farm1", "farm2"), times = 6),
  adg    = round(rnorm(12, 900, 50)),     # average daily gain
  bf     = round(rnorm(12, 11, 1.5), 1)   # backfat
)

# Example 1: column names
r2p_nest(
  growth,
  names_from = "farm",           # levels become columns: farm1, farm2
  cols      = c("adg", "bf"),    # traits to pivot
  id        = "animal"           # aligns records of the same animal
)
#>      name              data
#>    <char>            <list>
#> 1:    adg <data.table[6x3]>
#> 2:     bf <data.table[6x3]>
# Returns a nested data.table where:
# - name: trait names (adg, bf)
# - data: one row per animal with columns animal, farm1, farm2

# Example 2: numeric indices
r2p_nest(growth, names_from = 2, cols = 3:4, id = 1)
#>      name              data
#>    <char>            <list>
#> 1:    adg <data.table[6x3]>
#> 2:     bf <data.table[6x3]>

export_nest

# Example: Basic nested data export workflow
# A dedicated sub-folder of tempdir() keeps the clean-up safe
out_dir <- file.path(tempdir(), "mintyr_export_nest")

# Step 1: Create nested data structure
dt_nest <- w2l_nest(
  data = iris,              # Input iris dataset
  cols = 1:2,               # Columns to be nested
  by = "Species"            # Grouping variable
)

# Step 2: Export nested data to files
files <- export_nest(
  data = dt_nest,                    # Input nested data.table
  cols = "data",                     # Column containing nested data
  by = c("name", "Species"),         # Columns to create directory structure
  path = out_dir
)
#> [ export_nest ] Export complete. 6 file(s) written to: /tmp/RtmpuLygMH/mintyr_export_nest
# Returns (invisibly) the paths of the written files
# Directory structure: out_dir/<name>/<Species>/data.txt
files
#> [1] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Length/setosa/data.txt"    
#> [2] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Length/versicolor/data.txt"
#> [3] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Length/virginica/data.txt" 
#> [4] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Width/setosa/data.txt"     
#> [5] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Width/versicolor/data.txt" 
#> [6] "/tmp/RtmpuLygMH/mintyr_export_nest/Sepal.Width/virginica/data.txt"

# Clean up
unlink(out_dir, recursive = TRUE)

export_list

# Example: Export split data to files
out_dir <- file.path(tempdir(), "mintyr_export_list")

# Step 1: Create split data structure
dt_split <- w2l_split(
  data = iris,              # Input iris dataset
  cols = 1:2,               # Columns to be split
  by = "Species"            # Grouping variable
)

# Step 2: Export split data to files
files <- export_list(
  data = dt_split,          # Input list of data.tables
  path = out_dir
)
#> [ export_list ] Export complete. 6 / 6 file(s) written to: /tmp/RtmpuLygMH/mintyr_export_list
# Returns (invisibly) a named vector of the written file paths
files
#>                                              Sepal.Length_setosa 
#>     "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Length_setosa.txt" 
#>                                          Sepal.Length_versicolor 
#> "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Length_versicolor.txt" 
#>                                           Sepal.Length_virginica 
#>  "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Length_virginica.txt" 
#>                                               Sepal.Width_setosa 
#>      "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Width_setosa.txt" 
#>                                           Sepal.Width_versicolor 
#>  "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Width_versicolor.txt" 
#>                                            Sepal.Width_virginica 
#>   "/tmp/RtmpuLygMH/mintyr_export_list/Sepal.Width_virginica.txt"

# Clean up
unlink(out_dir, recursive = TRUE)

import_csv

# Example: CSV file import demonstrations

# Setup test files
csv_files <- mintyr_example(
  mintyr_examples("csv_test")     # Get example CSV files
)

# Example 1: Import and combine CSV files using data.table
import_csv(
  csv_files,                      # Input CSV file paths
  combine = TRUE,                 # Combine all files into one data.table
  file_col = "_file",             # Column name for file source
  keep_ext = TRUE,                # Include .csv extension in _file column
  full_path = TRUE                # Show complete file paths in _file column
)
#>                                                           _file  col1   col2
#>                                                          <char> <int> <char>
#> 1: /home/runner/work/_temp/Library/mintyr/extdata/csv_test1.csv     4      d
#> 2: /home/runner/work/_temp/Library/mintyr/extdata/csv_test1.csv     5      f
#> 3: /home/runner/work/_temp/Library/mintyr/extdata/csv_test1.csv     6      e
#> 4: /home/runner/work/_temp/Library/mintyr/extdata/csv_test2.csv    15      o
#> 5: /home/runner/work/_temp/Library/mintyr/extdata/csv_test2.csv    16      p
#> 6: /home/runner/work/_temp/Library/mintyr/extdata/csv_test2.csv    17      q
#>      col3
#>    <lgcl>
#> 1:  FALSE
#> 2:   TRUE
#> 3:   TRUE
#> 4:  FALSE
#> 5:   TRUE
#> 6:  FALSE

import_xlsx

# Example: Excel file import demonstrations

# Setup test files
xlsx_files <- mintyr_example(
  mintyr_examples("xlsx_test")    # Get example Excel files
)

# Example 1: Import and combine all sheets from all files
import_xlsx(
  xlsx_files,                     # Input Excel file paths
  combine = TRUE                  # Combine all sheets into one data.table
)
#>     excel_name sheet_name  col1   col2   col3
#>         <char>     <char> <num> <char> <lgcl>
#>  1: xlsx_test1     Sheet1     4      d  FALSE
#>  2: xlsx_test1     Sheet1     5      f   TRUE
#>  3: xlsx_test1     Sheet1     6      e   TRUE
#>  4: xlsx_test1     Sheet2     1      a   TRUE
#>  5: xlsx_test1     Sheet2     2      b  FALSE
#>  6: xlsx_test1     Sheet2     3      c   TRUE
#>  7: xlsx_test2     Sheet1    15      o  FALSE
#>  8: xlsx_test2     Sheet1    16      p   TRUE
#>  9: xlsx_test2     Sheet1    17      q  FALSE
#> 10: xlsx_test2          a     7      g  FALSE
#> 11: xlsx_test2          a     9      h   TRUE
#> 12: xlsx_test2          a     8      i  FALSE
#> 13: xlsx_test2          b    10      J  FALSE
#> 14: xlsx_test2          b    11      K   TRUE
#> 15: xlsx_test2          b    12      L  FALSE

# Example 2: Import specific sheets separately
import_xlsx(
  xlsx_files,                     # Input Excel file paths
  combine = FALSE,                 # Keep sheets as separate data.tables
  sheet = 2                       # Only import the second sheet
)
#> $xlsx_test1_Sheet2
#>     col1   col2   col3
#>    <num> <char> <lgcl>
#> 1:     1      a   TRUE
#> 2:     2      b  FALSE
#> 3:     3      c   TRUE
#> 
#> $xlsx_test2_a
#>     col1   col2   col3
#>    <num> <char> <lgcl>
#> 1:     7      g  FALSE
#> 2:     9      h   TRUE
#> 3:     8      i  FALSE
#> 
#> attr(,"source_files")
#> [1] "/home/runner/work/_temp/Library/mintyr/extdata/xlsx_test1.xlsx"
#> [2] "/home/runner/work/_temp/Library/mintyr/extdata/xlsx_test2.xlsx"

# Example 3: Multi-row header with merged cells and a title row
# (row 1 = title, rows 2-3 = header, data from row 4)
mh_file <- mintyr_example("multiheader_test.xlsx")
import_xlsx(
  mh_file,
  skip = 1,                       # Skip the title row
  header_rows = 2                 # Combine two header rows into one name
)
#>          excel_name sheet_name     ID     Breed Birth date Weight (kg)_Start
#>              <char>     <char> <char>    <char>     <POSc>             <num>
#> 1: multiheader_test     farm_A   A001     Duroc 2024-01-05              28.5
#> 2: multiheader_test     farm_A   A002     Duroc 2024-01-07              30.1
#> 3: multiheader_test     farm_A   A003  Landrace 2024-01-10              27.9
#> 4: multiheader_test     farm_A   A004  Landrace 2024-01-12              29.4
#> 5: multiheader_test     farm_B   B001 Yorkshire 2024-02-01              29.0
#> 6: multiheader_test     farm_B   B002 Yorkshire 2024-02-03              27.6
#> 7: multiheader_test     farm_B   B003     Duroc 2024-02-04              31.2
#>    Weight (kg)_End Backfat (mm)_P2 Backfat (mm)_Loin   Note
#>              <num>           <num>             <num> <char>
#> 1:           102.3            11.2              56.1   <NA>
#> 2:           108.7            12.5              58.3   lame
#> 3:            99.5            10.8              54.7   <NA>
#> 4:           104.2            11.9              57.0   <NA>
#> 5:           101.8            12.1              55.4   <NA>
#> 6:            98.4            10.9              53.9   <NA>
#> 7:           110.5            13.0              59.2 culled
# The last column has an empty, unmerged upper cell: it is named "Note".
# The heuristic fill would wrongly call it "Backfat (mm)_Note":
names(import_xlsx(mh_file, sheet = 1, skip = 1, header_rows = 2,
                  header_fill = "right"))
#>  [1] "excel_name"        "sheet_name"        "ID"               
#>  [4] "Breed"             "Birth date"        "Weight (kg)_Start"
#>  [7] "Weight (kg)_End"   "Backfat (mm)_P2"   "Backfat (mm)_Loin"
#> [10] "Backfat (mm)_Note"

# Example 4: Batch import that skips unreadable files instead of stopping
bad_file <- tempfile(fileext = ".xlsx")
writeLines("not a workbook", bad_file)
res <- suppressWarnings(
  import_xlsx(c(mh_file, bad_file), skip = 1, header_rows = 2, on_error = "warn")
)
unique(res$excel_name)
#> [1] "multiheader_test"
unlink(bad_file)

export_xlsx

# Example 1: A plain data.frame -> one workbook, one sheet
out_file <- file.path(tempdir(), "mtcars.xlsx")
export_xlsx(mtcars, path = out_file, sheet_name = "mtcars")
invisible(file.remove(out_file))

# Example 2: data.table input works exactly the same way
out_file <- file.path(tempdir(), "mtcars_dt.xlsx")
export_xlsx(data.table::as.data.table(mtcars),
            path = out_file, sheet_name = "mtcars")
invisible(file.remove(out_file))

# Example 3: One sheet per group in a single workbook
# Each Species value becomes a sheet; keep the Species column
out_file <- file.path(tempdir(), "iris_by_species.xlsx")
export_xlsx(iris, path = out_file, file_col = "Species", drop_cols = FALSE)
invisible(file.remove(out_file))

# Example 4: One file per group (directory mode: no .xlsx extension)
out_dir <- file.path(tempdir(), "iris_by_species")
out_files <- export_xlsx(iris, path = out_dir, file_col = "Species")
basename(out_files)
#> [1] "setosa.xlsx"     "versicolor.xlsx" "virginica.xlsx"
unlink(out_dir, recursive = TRUE)

# Example 5: Round-trip the layout produced by import_xlsx(combine = TRUE)
# Rows are routed back to their original file and sheet
combined <- data.frame(
  excel_name = c("sales", "sales", "costs"),
  sheet_name = c("2024",  "2025",  "2024"),
  amount     = c(100, 120, 80)
)
out_dir <- file.path(tempdir(), "roundtrip")
out_files <- export_xlsx(combined, path = out_dir)   # sales.xlsx, costs.xlsx
basename(out_files)
#> [1] "sales.xlsx" "costs.xlsx"
unlink(out_dir, recursive = TRUE)

# Example 6: Named list -> single workbook, one sheet per element
out_file <- file.path(tempdir(), "combined.xlsx")
res <- list(res1 = iris, res2 = mtcars)
export_xlsx(res, path = out_file)
invisible(file.remove(out_file))

# Example 7: Named list -> directory, one file per element
out_dir <- file.path(tempdir(), "combined")
out_files <- export_xlsx(res, path = out_dir)        # res1.xlsx, res2.xlsx
basename(out_files)
#> [1] "res1.xlsx" "res2.xlsx"
unlink(out_dir, recursive = TRUE)

top_perc

# Example 1: Basic usage with single trait
# This example selects the top 10% of observations based on Petal.Width
# keep_data=TRUE returns both summary statistics and the filtered data
top_perc(iris, 
         perc = 0.1,                # Select top 10%
         cols = c("Petal.Width"),   # Column to analyze
         keep_data = TRUE)          # Return both stats and filtered data
#> $perc_0.1
#> $perc_0.1$stat
#>      variable  n min max     mean median         sd         se         cv
#> 1 Petal.Width 17 2.2 2.5 2.335294    2.3 0.09963167 0.02416423 0.04266344
#>   selection
#> 1   top_10%
#> 
#> $perc_0.1$data
#>    Sepal.Length Sepal.Width Petal.Length   Species    variable value
#> 1           6.3         3.3          6.0 virginica Petal.Width   2.5
#> 2           6.5         3.0          5.8 virginica Petal.Width   2.2
#> 3           7.2         3.6          6.1 virginica Petal.Width   2.5
#> 4           5.8         2.8          5.1 virginica Petal.Width   2.4
#> 5           6.4         3.2          5.3 virginica Petal.Width   2.3
#> 6           7.7         3.8          6.7 virginica Petal.Width   2.2
#> 7           7.7         2.6          6.9 virginica Petal.Width   2.3
#> 8           6.9         3.2          5.7 virginica Petal.Width   2.3
#> 9           6.4         2.8          5.6 virginica Petal.Width   2.2
#> 10          7.7         3.0          6.1 virginica Petal.Width   2.3
#> 11          6.3         3.4          5.6 virginica Petal.Width   2.4
#> 12          6.7         3.1          5.6 virginica Petal.Width   2.4
#> 13          6.9         3.1          5.1 virginica Petal.Width   2.3
#> 14          6.8         3.2          5.9 virginica Petal.Width   2.3
#> 15          6.7         3.3          5.7 virginica Petal.Width   2.5
#> 16          6.7         3.0          5.2 virginica Petal.Width   2.3
#> 17          6.2         3.4          5.4 virginica Petal.Width   2.3

# Example 2: Using grouping with 'by' parameter
# This example performs the same analysis but separately for each Species
# Returns a data.frame of summary statistics, one row per Species
top_perc(iris, 
         perc = 0.1,                # Select top 10%
         cols = c("Petal.Width"),   # Column to analyze
         by = "Species")            # Group by Species
#>      Species    variable n min max      mean median         sd         se
#> 1     setosa Petal.Width 9 0.4 0.6 0.4333333   0.40 0.07071068 0.02357023
#> 2 versicolor Petal.Width 5 1.6 1.8 1.6600000   1.60 0.08944272 0.04000000
#> 3  virginica Petal.Width 6 2.4 2.5 2.4500000   2.45 0.05477226 0.02236068
#>           cv selection
#> 1 0.16317849   top_10%
#> 2 0.05388116   top_10%
#> 3 0.02235602   top_10%

desc_stats

# Example 1: Common statistics for every numeric column of iris
desc_stats(iris)
#>        variable     n   min   max median   iqr     mean        sd         se
#>          <char> <int> <num> <num>  <num> <num>    <num>     <num>      <num>
#> 1: Sepal.Length   150   4.3   7.9   5.80   1.3 5.843333 0.8280661 0.06761132
#> 2:  Sepal.Width   150   2.0   4.4   3.00   0.5 3.057333 0.4358663 0.03558833
#> 3: Petal.Length   150   1.0   6.9   4.35   3.5 3.758000 1.7652982 0.14413600
#> 4:  Petal.Width   150   0.1   2.5   1.30   1.5 1.199333 0.7622377 0.06223645
#>            ci
#>         <num>
#> 1: 0.13360085
#> 2: 0.07032302
#> 3: 0.28481463
#> 4: 0.12298004

# Example 2: Mean and SD by group, with a total row over all records
desc_stats(
  iris,
  by = "Species",                 # Grouping column
  type = "mean_sd",               # Preset: n, mean, sd
  total = TRUE,                   # n = sum of the groups, mean / sd of all records
  digits = 2                      # Round the statistics
)
#>        Species     variable     n  mean    sd
#>         <fctr>       <char> <int> <num> <num>
#>  1:     setosa Sepal.Length    50  5.01  0.35
#>  2: versicolor Sepal.Length    50  5.94  0.52
#>  3:  virginica Sepal.Length    50  6.59  0.64
#>  4:      Total Sepal.Length   150  5.84  0.83
#>  5:     setosa  Sepal.Width    50  3.43  0.38
#>  6: versicolor  Sepal.Width    50  2.77  0.31
#>  7:  virginica  Sepal.Width    50  2.97  0.32
#>  8:      Total  Sepal.Width   150  3.06  0.44
#>  9:     setosa Petal.Length    50  1.46  0.17
#> 10: versicolor Petal.Length    50  4.26  0.47
#> 11:  virginica Petal.Length    50  5.55  0.55
#> 12:      Total Petal.Length   150  3.76  1.77
#> 13:     setosa  Petal.Width    50  0.25  0.11
#> 14: versicolor  Petal.Width    50  1.33  0.20
#> 15:  virginica  Petal.Width    50  2.03  0.27
#> 16:      Total  Petal.Width   150  1.20  0.76
#> 17:     setosa        Total   200    NA    NA
#> 18: versicolor        Total   200    NA    NA
#> 19:  virginica        Total   200    NA    NA
#> 20:      Total        Total   600    NA    NA
#>        Species     variable     n  mean    sd
#>         <fctr>       <char> <int> <num> <num>

# Example 3: Count table with row and column totals
# (counts are the only statistic that adds up across variables)
desc_stats(
  iris,
  by = "Species",
  stats = "n",
  total = TRUE,                   # Total row and Total column
  shape = "wide"
)
#>       Species Sepal.Length Sepal.Width Petal.Length Petal.Width Total
#>        <fctr>        <int>       <int>        <int>       <int> <int>
#> 1:     setosa           50          50           50          50   200
#> 2: versicolor           50          50           50          50   200
#> 3:  virginica           50          50           50          50   200
#> 4:      Total          150         150          150         150   600

# Without groups the total adds up the counts of all variables
desc_stats(iris, type = "mean_sd", total = TRUE, digits = 2)
#>        variable     n  mean    sd
#>          <char> <int> <num> <num>
#> 1: Sepal.Length   150  5.84  0.83
#> 2:  Sepal.Width   150  3.06  0.44
#> 3: Petal.Length   150  3.76  1.77
#> 4:  Petal.Width   150  1.20  0.76
#> 5:        Total   600    NA    NA

# Example 3b: Report table - groups in rows, traits in columns, "mean ± sd"
desc_stats(
  mtcars,
  cols = c("mpg", "hp", "wt"),
  by = "cyl",
  fmt = "{mean} ± {sd}",          # Report-ready text
  total = TRUE,
  shape = "wide"                  # One row per group, one column per trait
)
#>       cyl          mpg             hp          wt
#>    <char>       <char>         <char>      <char>
#> 1:      4 26.66 ± 4.51  82.64 ± 20.93 2.29 ± 0.57
#> 2:      6 19.74 ± 1.45 122.29 ± 24.26 3.12 ± 0.36
#> 3:      8 15.10 ± 2.56 209.21 ± 50.98 4.00 ± 0.76
#> 4:  Total 20.09 ± 6.03 146.69 ± 68.56 3.22 ± 0.98

# Example 4: Several templates, per-placeholder decimals, CV in percent and
# percentiles
desc_stats(
  mtcars,
  cols = c("mpg", "wt"),
  by = "am",
  fmt = c(N = "{n}",
          "Mean ± SD" = "{mean:1} ± {sd:1}",
          "CV" = "{cv:1%}",
          "Median [P2.5, P97.5]" = "{median:1} [{q2.5:1}, {q97.5:1}]"),
  total = TRUE
)
#>        am variable      N  Mean ± SD     CV Median [P2.5, P97.5]
#>    <char>   <char> <char>     <char> <char>               <char>
#> 1:      0      mpg     19 17.1 ± 3.8  22.4%    17.3 [10.4, 23.7]
#> 2:      1      mpg     13 24.4 ± 6.2  25.3%    22.8 [15.2, 33.4]
#> 3:  Total      mpg     32 20.1 ± 6.0  30.0%    19.2 [10.4, 32.7]
#> 4:      0       wt     19  3.8 ± 0.8  20.6%       3.5 [2.8, 5.4]
#> 5:      1       wt     13  2.4 ± 0.6  25.6%       2.3 [1.5, 3.4]
#> 6:  Total       wt     32  3.2 ± 1.0  30.4%       3.3 [1.6, 5.4]
#> 7:      0    Total     38       <NA>   <NA>                 <NA>
#> 8:      1    Total     26       <NA>   <NA>                 <NA>
#> 9:  Total    Total     64       <NA>   <NA>                 <NA>

# Example 5: Quantiles, returned as a data.frame
desc_stats(iris, cols = 1:2, type = "quantile",
           probs = c(0.05, 0.5, 0.95), out_type = "df")
#>       variable   n    q5 q50   q95
#> 1 Sepal.Length 150 4.600 5.8 7.255
#> 2  Sepal.Width 150 2.345 3.0 3.800

get_path_info

paths <- c("C:/Users/foo/Documents/report.xlsx",
           "/home/user/.bashrc",
           "relative/path/to/data.csv",
           ".hidden.tar.gz",
           NA_character_)

# Mode B: filename only, extension stripped (default)
get_path_info(paths)
#> [1] "report"      ".bashrc"     "data"        ".hidden.tar" NA

# Mode B: filename only, extension preserved
get_path_info(paths, rm_extension = FALSE)
#> [1] "report.xlsx"    ".bashrc"        "data.csv"       ".hidden.tar.gz"
#> [5] NA

# Mode B: full normalised path, extension stripped
get_path_info(paths, rm_path = FALSE)
#> [1] "C:/Users/foo/Documents/report" "/home/user/.bashrc"           
#> [3] "relative/path/to/data"         ".hidden.tar"                  
#> [5] NA

# Mode A: extract the 2nd path segment
get_path_info(paths, n = 2)
#> [1] "foo"  "user" "path" NA     NA

# Mode A: extract the last segment with extension stripped (n = -1 linkage)
get_path_info(paths, n = -1, rm_extension = TRUE)
#> [1] "report"      ".bashrc"     "data"        ".hidden.tar" NA

# Mode A: range extraction
get_path_info(paths, n = c(2, 3))
#> [1] "foo/Documents" "user/.bashrc"  "path/to"       NA             
#> [5] NA

format_digits

# Example: Number formatting demonstrations

# Setup test data
dt <- data.table::data.table(
  a = c(0.1234, 0.5678),      # Numeric column 1
  b = c(0.2345, 0.6789),      # Numeric column 2
  c = c("text1", "text2")     # Text column
)

# Example 1: Format all numeric columns
format_digits(
  dt,                         # Input data table
  digits = 2                  # Round to 2 decimal places
)
#>         a      b      c
#>    <char> <char> <char>
#> 1:   0.12   0.23  text1
#> 2:   0.57   0.68  text2

# Example 2: Format specific column as percentage
format_digits(
  dt,                         # Input data table
  cols = c("a"),              # Only format column 'a'
  digits = 2,                 # Round to 2 decimal places
  percentage = TRUE           # Convert to percentage
)
#>         a      b      c
#>    <char>  <num> <char>
#> 1: 12.34% 0.2345  text1
#> 2: 56.78% 0.6789  text2

mintyr_example

# Get path to an example file
mintyr_example("csv_test1.csv")
#> [1] "/home/runner/work/_temp/Library/mintyr/extdata/csv_test1.csv"

mintyr_examples

# List all example files
mintyr_examples()
#> [1] "csv_test1.csv"         "csv_test2.csv"         "multiheader_test.xlsx"
#> [4] "xlsx_test1.xlsx"       "xlsx_test2.xlsx"