Descriptive statistics

library(mintyr)

desc_stats() summarises numeric columns by group, with optional totals, report-ready text ("{mean} ± {sd}") and a wide report layout. top_perc() describes the best (or worst) X% of records per group, and format_digits() formats numbers for tables.

Descriptive statistics by group: desc_stats()

# Example 1: Common statistics for every numeric column of iris
desc_stats(iris)
#>        variable     n   min   max median   iqr     mean        sd         se
#>          <char> <int> <num> <num>  <num> <num>    <num>     <num>      <num>
#> 1: Sepal.Length   150   4.3   7.9   5.80   1.3 5.843333 0.8280661 0.06761132
#> 2:  Sepal.Width   150   2.0   4.4   3.00   0.5 3.057333 0.4358663 0.03558833
#> 3: Petal.Length   150   1.0   6.9   4.35   3.5 3.758000 1.7652982 0.14413600
#> 4:  Petal.Width   150   0.1   2.5   1.30   1.5 1.199333 0.7622377 0.06223645
#>            ci
#>         <num>
#> 1: 0.13360085
#> 2: 0.07032302
#> 3: 0.28481463
#> 4: 0.12298004

# Example 2: Mean and SD by group, with a total row over all records
desc_stats(
  iris,
  by = "Species",                 # Grouping column
  type = "mean_sd",               # Preset: n, mean, sd
  total = TRUE,                   # n = sum of the groups, mean / sd of all records
  digits = 2                      # Round the statistics
)
#>        Species     variable     n  mean    sd
#>         <fctr>       <char> <int> <num> <num>
#>  1:     setosa Sepal.Length    50  5.01  0.35
#>  2: versicolor Sepal.Length    50  5.94  0.52
#>  3:  virginica Sepal.Length    50  6.59  0.64
#>  4:      Total Sepal.Length   150  5.84  0.83
#>  5:     setosa  Sepal.Width    50  3.43  0.38
#>  6: versicolor  Sepal.Width    50  2.77  0.31
#>  7:  virginica  Sepal.Width    50  2.97  0.32
#>  8:      Total  Sepal.Width   150  3.06  0.44
#>  9:     setosa Petal.Length    50  1.46  0.17
#> 10: versicolor Petal.Length    50  4.26  0.47
#> 11:  virginica Petal.Length    50  5.55  0.55
#> 12:      Total Petal.Length   150  3.76  1.77
#> 13:     setosa  Petal.Width    50  0.25  0.11
#> 14: versicolor  Petal.Width    50  1.33  0.20
#> 15:  virginica  Petal.Width    50  2.03  0.27
#> 16:      Total  Petal.Width   150  1.20  0.76
#> 17:     setosa        Total   200    NA    NA
#> 18: versicolor        Total   200    NA    NA
#> 19:  virginica        Total   200    NA    NA
#> 20:      Total        Total   600    NA    NA
#>        Species     variable     n  mean    sd
#>         <fctr>       <char> <int> <num> <num>

# Example 3: Count table with row and column totals
# (counts are the only statistic that adds up across variables)
desc_stats(
  iris,
  by = "Species",
  stats = "n",
  total = TRUE,                   # Total row and Total column
  shape = "wide"
)
#>       Species Sepal.Length Sepal.Width Petal.Length Petal.Width Total
#>        <fctr>        <int>       <int>        <int>       <int> <int>
#> 1:     setosa           50          50           50          50   200
#> 2: versicolor           50          50           50          50   200
#> 3:  virginica           50          50           50          50   200
#> 4:      Total          150         150          150         150   600

# Without groups the total adds up the counts of all variables
desc_stats(iris, type = "mean_sd", total = TRUE, digits = 2)
#>        variable     n  mean    sd
#>          <char> <int> <num> <num>
#> 1: Sepal.Length   150  5.84  0.83
#> 2:  Sepal.Width   150  3.06  0.44
#> 3: Petal.Length   150  3.76  1.77
#> 4:  Petal.Width   150  1.20  0.76
#> 5:        Total   600    NA    NA

# Example 3b: Report table - groups in rows, traits in columns, "mean ± sd"
desc_stats(
  mtcars,
  cols = c("mpg", "hp", "wt"),
  by = "cyl",
  fmt = "{mean} ± {sd}",          # Report-ready text
  total = TRUE,
  shape = "wide"                  # One row per group, one column per trait
)
#>       cyl          mpg             hp          wt
#>    <char>       <char>         <char>      <char>
#> 1:      4 26.66 ± 4.51  82.64 ± 20.93 2.29 ± 0.57
#> 2:      6 19.74 ± 1.45 122.29 ± 24.26 3.12 ± 0.36
#> 3:      8 15.10 ± 2.56 209.21 ± 50.98 4.00 ± 0.76
#> 4:  Total 20.09 ± 6.03 146.69 ± 68.56 3.22 ± 0.98

# Example 4: Several templates, per-placeholder decimals, CV in percent and
# percentiles
desc_stats(
  mtcars,
  cols = c("mpg", "wt"),
  by = "am",
  fmt = c(N = "{n}",
          "Mean ± SD" = "{mean:1} ± {sd:1}",
          "CV" = "{cv:1%}",
          "Median [P2.5, P97.5]" = "{median:1} [{q2.5:1}, {q97.5:1}]"),
  total = TRUE
)
#>        am variable      N  Mean ± SD     CV Median [P2.5, P97.5]
#>    <char>   <char> <char>     <char> <char>               <char>
#> 1:      0      mpg     19 17.1 ± 3.8  22.4%    17.3 [10.4, 23.7]
#> 2:      1      mpg     13 24.4 ± 6.2  25.3%    22.8 [15.2, 33.4]
#> 3:  Total      mpg     32 20.1 ± 6.0  30.0%    19.2 [10.4, 32.7]
#> 4:      0       wt     19  3.8 ± 0.8  20.6%       3.5 [2.8, 5.4]
#> 5:      1       wt     13  2.4 ± 0.6  25.6%       2.3 [1.5, 3.4]
#> 6:  Total       wt     32  3.2 ± 1.0  30.4%       3.3 [1.6, 5.4]
#> 7:      0    Total     38       <NA>   <NA>                 <NA>
#> 8:      1    Total     26       <NA>   <NA>                 <NA>
#> 9:  Total    Total     64       <NA>   <NA>                 <NA>

# Example 5: Quantiles, returned as a data.frame
desc_stats(iris, cols = 1:2, type = "quantile",
           probs = c(0.05, 0.5, 0.95), out_type = "df")
#>       variable   n    q5 q50   q95
#> 1 Sepal.Length 150 4.600 5.8 7.255
#> 2  Sepal.Width 150 2.345 3.0 3.800

Top / bottom X% per group: top_perc()

# Example 1: Basic usage with single trait
# This example selects the top 10% of observations based on Petal.Width
# keep_data=TRUE returns both summary statistics and the filtered data
top_perc(iris, 
         perc = 0.1,                # Select top 10%
         cols = c("Petal.Width"),   # Column to analyze
         keep_data = TRUE)          # Return both stats and filtered data
#> $perc_0.1
#> $perc_0.1$stat
#>      variable  n min max     mean median         sd         se         cv
#> 1 Petal.Width 17 2.2 2.5 2.335294    2.3 0.09963167 0.02416423 0.04266344
#>   selection
#> 1   top_10%
#> 
#> $perc_0.1$data
#>    Sepal.Length Sepal.Width Petal.Length   Species    variable value
#> 1           6.3         3.3          6.0 virginica Petal.Width   2.5
#> 2           6.5         3.0          5.8 virginica Petal.Width   2.2
#> 3           7.2         3.6          6.1 virginica Petal.Width   2.5
#> 4           5.8         2.8          5.1 virginica Petal.Width   2.4
#> 5           6.4         3.2          5.3 virginica Petal.Width   2.3
#> 6           7.7         3.8          6.7 virginica Petal.Width   2.2
#> 7           7.7         2.6          6.9 virginica Petal.Width   2.3
#> 8           6.9         3.2          5.7 virginica Petal.Width   2.3
#> 9           6.4         2.8          5.6 virginica Petal.Width   2.2
#> 10          7.7         3.0          6.1 virginica Petal.Width   2.3
#> 11          6.3         3.4          5.6 virginica Petal.Width   2.4
#> 12          6.7         3.1          5.6 virginica Petal.Width   2.4
#> 13          6.9         3.1          5.1 virginica Petal.Width   2.3
#> 14          6.8         3.2          5.9 virginica Petal.Width   2.3
#> 15          6.7         3.3          5.7 virginica Petal.Width   2.5
#> 16          6.7         3.0          5.2 virginica Petal.Width   2.3
#> 17          6.2         3.4          5.4 virginica Petal.Width   2.3

# Example 2: Using grouping with 'by' parameter
# This example performs the same analysis but separately for each Species
# Returns a data.frame of summary statistics, one row per Species
top_perc(iris, 
         perc = 0.1,                # Select top 10%
         cols = c("Petal.Width"),   # Column to analyze
         by = "Species")            # Group by Species
#>      Species    variable n min max      mean median         sd         se
#> 1     setosa Petal.Width 9 0.4 0.6 0.4333333   0.40 0.07071068 0.02357023
#> 2 versicolor Petal.Width 5 1.6 1.8 1.6600000   1.60 0.08944272 0.04000000
#> 3  virginica Petal.Width 6 2.4 2.5 2.4500000   2.45 0.05477226 0.02236068
#>           cv selection
#> 1 0.16317849   top_10%
#> 2 0.05388116   top_10%
#> 3 0.02235602   top_10%

Format numbers for reports: format_digits()

# Example: Number formatting demonstrations

# Setup test data
dt <- data.table::data.table(
  a = c(0.1234, 0.5678),      # Numeric column 1
  b = c(0.2345, 0.6789),      # Numeric column 2
  c = c("text1", "text2")     # Text column
)

# Example 1: Format all numeric columns
format_digits(
  dt,                         # Input data table
  digits = 2                  # Round to 2 decimal places
)
#>         a      b      c
#>    <char> <char> <char>
#> 1:   0.12   0.23  text1
#> 2:   0.57   0.68  text2

# Example 2: Format specific column as percentage
format_digits(
  dt,                         # Input data table
  cols = c("a"),              # Only format column 'a'
  digits = 2,                 # Round to 2 decimal places
  percentage = TRUE           # Convert to percentage
)
#>         a      b      c
#>    <char>  <num> <char>
#> 1: 12.34% 0.2345  text1
#> 2: 56.78% 0.6789  text2