desc_stats() summarises numeric columns by group, with
optional totals, report-ready text ("{mean} ± {sd}") and a
wide report layout. top_perc() describes the best (or
worst) X% of records per group, and format_digits() formats
numbers for tables.
desc_stats()
# Example 1: Common statistics for every numeric column of iris
desc_stats(iris)
#> variable n min max median iqr mean sd se
#> <char> <int> <num> <num> <num> <num> <num> <num> <num>
#> 1: Sepal.Length 150 4.3 7.9 5.80 1.3 5.843333 0.8280661 0.06761132
#> 2: Sepal.Width 150 2.0 4.4 3.00 0.5 3.057333 0.4358663 0.03558833
#> 3: Petal.Length 150 1.0 6.9 4.35 3.5 3.758000 1.7652982 0.14413600
#> 4: Petal.Width 150 0.1 2.5 1.30 1.5 1.199333 0.7622377 0.06223645
#> ci
#> <num>
#> 1: 0.13360085
#> 2: 0.07032302
#> 3: 0.28481463
#> 4: 0.12298004
# Example 2: Mean and SD by group, with a total row over all records
desc_stats(
iris,
by = "Species", # Grouping column
type = "mean_sd", # Preset: n, mean, sd
total = TRUE, # n = sum of the groups, mean / sd of all records
digits = 2 # Round the statistics
)
#> Species variable n mean sd
#> <fctr> <char> <int> <num> <num>
#> 1: setosa Sepal.Length 50 5.01 0.35
#> 2: versicolor Sepal.Length 50 5.94 0.52
#> 3: virginica Sepal.Length 50 6.59 0.64
#> 4: Total Sepal.Length 150 5.84 0.83
#> 5: setosa Sepal.Width 50 3.43 0.38
#> 6: versicolor Sepal.Width 50 2.77 0.31
#> 7: virginica Sepal.Width 50 2.97 0.32
#> 8: Total Sepal.Width 150 3.06 0.44
#> 9: setosa Petal.Length 50 1.46 0.17
#> 10: versicolor Petal.Length 50 4.26 0.47
#> 11: virginica Petal.Length 50 5.55 0.55
#> 12: Total Petal.Length 150 3.76 1.77
#> 13: setosa Petal.Width 50 0.25 0.11
#> 14: versicolor Petal.Width 50 1.33 0.20
#> 15: virginica Petal.Width 50 2.03 0.27
#> 16: Total Petal.Width 150 1.20 0.76
#> 17: setosa Total 200 NA NA
#> 18: versicolor Total 200 NA NA
#> 19: virginica Total 200 NA NA
#> 20: Total Total 600 NA NA
#> Species variable n mean sd
#> <fctr> <char> <int> <num> <num>
# Example 3: Count table with row and column totals
# (counts are the only statistic that adds up across variables)
desc_stats(
iris,
by = "Species",
stats = "n",
total = TRUE, # Total row and Total column
shape = "wide"
)
#> Species Sepal.Length Sepal.Width Petal.Length Petal.Width Total
#> <fctr> <int> <int> <int> <int> <int>
#> 1: setosa 50 50 50 50 200
#> 2: versicolor 50 50 50 50 200
#> 3: virginica 50 50 50 50 200
#> 4: Total 150 150 150 150 600
# Without groups the total adds up the counts of all variables
desc_stats(iris, type = "mean_sd", total = TRUE, digits = 2)
#> variable n mean sd
#> <char> <int> <num> <num>
#> 1: Sepal.Length 150 5.84 0.83
#> 2: Sepal.Width 150 3.06 0.44
#> 3: Petal.Length 150 3.76 1.77
#> 4: Petal.Width 150 1.20 0.76
#> 5: Total 600 NA NA
# Example 3b: Report table - groups in rows, traits in columns, "mean ± sd"
desc_stats(
mtcars,
cols = c("mpg", "hp", "wt"),
by = "cyl",
fmt = "{mean} ± {sd}", # Report-ready text
total = TRUE,
shape = "wide" # One row per group, one column per trait
)
#> cyl mpg hp wt
#> <char> <char> <char> <char>
#> 1: 4 26.66 ± 4.51 82.64 ± 20.93 2.29 ± 0.57
#> 2: 6 19.74 ± 1.45 122.29 ± 24.26 3.12 ± 0.36
#> 3: 8 15.10 ± 2.56 209.21 ± 50.98 4.00 ± 0.76
#> 4: Total 20.09 ± 6.03 146.69 ± 68.56 3.22 ± 0.98
# Example 4: Several templates, per-placeholder decimals, CV in percent and
# percentiles
desc_stats(
mtcars,
cols = c("mpg", "wt"),
by = "am",
fmt = c(N = "{n}",
"Mean ± SD" = "{mean:1} ± {sd:1}",
"CV" = "{cv:1%}",
"Median [P2.5, P97.5]" = "{median:1} [{q2.5:1}, {q97.5:1}]"),
total = TRUE
)
#> am variable N Mean ± SD CV Median [P2.5, P97.5]
#> <char> <char> <char> <char> <char> <char>
#> 1: 0 mpg 19 17.1 ± 3.8 22.4% 17.3 [10.4, 23.7]
#> 2: 1 mpg 13 24.4 ± 6.2 25.3% 22.8 [15.2, 33.4]
#> 3: Total mpg 32 20.1 ± 6.0 30.0% 19.2 [10.4, 32.7]
#> 4: 0 wt 19 3.8 ± 0.8 20.6% 3.5 [2.8, 5.4]
#> 5: 1 wt 13 2.4 ± 0.6 25.6% 2.3 [1.5, 3.4]
#> 6: Total wt 32 3.2 ± 1.0 30.4% 3.3 [1.6, 5.4]
#> 7: 0 Total 38 <NA> <NA> <NA>
#> 8: 1 Total 26 <NA> <NA> <NA>
#> 9: Total Total 64 <NA> <NA> <NA>
# Example 5: Quantiles, returned as a data.frame
desc_stats(iris, cols = 1:2, type = "quantile",
probs = c(0.05, 0.5, 0.95), out_type = "df")
#> variable n q5 q50 q95
#> 1 Sepal.Length 150 4.600 5.8 7.255
#> 2 Sepal.Width 150 2.345 3.0 3.800top_perc()
# Example 1: Basic usage with single trait
# This example selects the top 10% of observations based on Petal.Width
# keep_data=TRUE returns both summary statistics and the filtered data
top_perc(iris,
perc = 0.1, # Select top 10%
cols = c("Petal.Width"), # Column to analyze
keep_data = TRUE) # Return both stats and filtered data
#> $perc_0.1
#> $perc_0.1$stat
#> variable n min max mean median sd se cv
#> 1 Petal.Width 17 2.2 2.5 2.335294 2.3 0.09963167 0.02416423 0.04266344
#> selection
#> 1 top_10%
#>
#> $perc_0.1$data
#> Sepal.Length Sepal.Width Petal.Length Species variable value
#> 1 6.3 3.3 6.0 virginica Petal.Width 2.5
#> 2 6.5 3.0 5.8 virginica Petal.Width 2.2
#> 3 7.2 3.6 6.1 virginica Petal.Width 2.5
#> 4 5.8 2.8 5.1 virginica Petal.Width 2.4
#> 5 6.4 3.2 5.3 virginica Petal.Width 2.3
#> 6 7.7 3.8 6.7 virginica Petal.Width 2.2
#> 7 7.7 2.6 6.9 virginica Petal.Width 2.3
#> 8 6.9 3.2 5.7 virginica Petal.Width 2.3
#> 9 6.4 2.8 5.6 virginica Petal.Width 2.2
#> 10 7.7 3.0 6.1 virginica Petal.Width 2.3
#> 11 6.3 3.4 5.6 virginica Petal.Width 2.4
#> 12 6.7 3.1 5.6 virginica Petal.Width 2.4
#> 13 6.9 3.1 5.1 virginica Petal.Width 2.3
#> 14 6.8 3.2 5.9 virginica Petal.Width 2.3
#> 15 6.7 3.3 5.7 virginica Petal.Width 2.5
#> 16 6.7 3.0 5.2 virginica Petal.Width 2.3
#> 17 6.2 3.4 5.4 virginica Petal.Width 2.3
# Example 2: Using grouping with 'by' parameter
# This example performs the same analysis but separately for each Species
# Returns a data.frame of summary statistics, one row per Species
top_perc(iris,
perc = 0.1, # Select top 10%
cols = c("Petal.Width"), # Column to analyze
by = "Species") # Group by Species
#> Species variable n min max mean median sd se
#> 1 setosa Petal.Width 9 0.4 0.6 0.4333333 0.40 0.07071068 0.02357023
#> 2 versicolor Petal.Width 5 1.6 1.8 1.6600000 1.60 0.08944272 0.04000000
#> 3 virginica Petal.Width 6 2.4 2.5 2.4500000 2.45 0.05477226 0.02236068
#> cv selection
#> 1 0.16317849 top_10%
#> 2 0.05388116 top_10%
#> 3 0.02235602 top_10%format_digits()
# Example: Number formatting demonstrations
# Setup test data
dt <- data.table::data.table(
a = c(0.1234, 0.5678), # Numeric column 1
b = c(0.2345, 0.6789), # Numeric column 2
c = c("text1", "text2") # Text column
)
# Example 1: Format all numeric columns
format_digits(
dt, # Input data table
digits = 2 # Round to 2 decimal places
)
#> a b c
#> <char> <char> <char>
#> 1: 0.12 0.23 text1
#> 2: 0.57 0.68 text2
# Example 2: Format specific column as percentage
format_digits(
dt, # Input data table
cols = c("a"), # Only format column 'a'
digits = 2, # Round to 2 decimal places
percentage = TRUE # Convert to percentage
)
#> a b c
#> <char> <num> <char>
#> 1: 12.34% 0.2345 text1
#> 2: 56.78% 0.6789 text2