---
title: "Descriptive statistics"
output: rmarkdown::html_vignette
vignette: >
  %\VignetteIndexEntry{Descriptive statistics}
  %\VignetteEngine{knitr::rmarkdown}
  %\VignetteEncoding{UTF-8}
---

```{r, include = FALSE}
knitr::opts_chunk$set(
  collapse = TRUE,
  comment = "#>"
)
```

```{r}
library(mintyr)
```

<!-- WARNING - This vignette is generated by {fusen} from dev/flat_summary.Rmd: do not edit by hand -->



`desc_stats()` summarises numeric columns by group, with optional totals, report-ready text (`"{mean} ± {sd}"`) and a wide report layout. `top_perc()` describes the best (or worst) X% of records per group, and `format_digits()` formats numbers for tables.

# Descriptive statistics by group: `desc_stats()`

```{r example-desc_stats}
# Example 1: Common statistics for every numeric column of iris
desc_stats(iris)

# Example 2: Mean and SD by group, with a total row over all records
desc_stats(
  iris,
  by = "Species",                 # Grouping column
  type = "mean_sd",               # Preset: n, mean, sd
  total = TRUE,                   # n = sum of the groups, mean / sd of all records
  digits = 2                      # Round the statistics
)

# Example 3: Count table with row and column totals
# (counts are the only statistic that adds up across variables)
desc_stats(
  iris,
  by = "Species",
  stats = "n",
  total = TRUE,                   # Total row and Total column
  shape = "wide"
)

# Without groups the total adds up the counts of all variables
desc_stats(iris, type = "mean_sd", total = TRUE, digits = 2)

# Example 3b: Report table - groups in rows, traits in columns, "mean ± sd"
desc_stats(
  mtcars,
  cols = c("mpg", "hp", "wt"),
  by = "cyl",
  fmt = "{mean} ± {sd}",          # Report-ready text
  total = TRUE,
  shape = "wide"                  # One row per group, one column per trait
)

# Example 4: Several templates, per-placeholder decimals, CV in percent and
# percentiles
desc_stats(
  mtcars,
  cols = c("mpg", "wt"),
  by = "am",
  fmt = c(N = "{n}",
          "Mean ± SD" = "{mean:1} ± {sd:1}",
          "CV" = "{cv:1%}",
          "Median [P2.5, P97.5]" = "{median:1} [{q2.5:1}, {q97.5:1}]"),
  total = TRUE
)

# Example 5: Quantiles, returned as a data.frame
desc_stats(iris, cols = 1:2, type = "quantile",
           probs = c(0.05, 0.5, 0.95), out_type = "df")
```



# Top / bottom X% per group: `top_perc()`
    
  
```{r example-top_perc}
# Example 1: Basic usage with single trait
# This example selects the top 10% of observations based on Petal.Width
# keep_data=TRUE returns both summary statistics and the filtered data
top_perc(iris, 
         perc = 0.1,                # Select top 10%
         cols = c("Petal.Width"),   # Column to analyze
         keep_data = TRUE)          # Return both stats and filtered data

# Example 2: Using grouping with 'by' parameter
# This example performs the same analysis but separately for each Species
# Returns a data.frame of summary statistics, one row per Species
top_perc(iris, 
         perc = 0.1,                # Select top 10%
         cols = c("Petal.Width"),   # Column to analyze
         by = "Species")            # Group by Species
```



# Format numbers for reports: `format_digits()`
    
  
```{r example-format_digits}
# Example: Number formatting demonstrations

# Setup test data
dt <- data.table::data.table(
  a = c(0.1234, 0.5678),      # Numeric column 1
  b = c(0.2345, 0.6789),      # Numeric column 2
  c = c("text1", "text2")     # Text column
)

# Example 1: Format all numeric columns
format_digits(
  dt,                         # Input data table
  digits = 2                  # Round to 2 decimal places
)

# Example 2: Format specific column as percentage
format_digits(
  dt,                         # Input data table
  cols = c("a"),              # Only format column 'a'
  digits = 2,                 # Round to 2 decimal places
  percentage = TRUE           # Convert to percentage
)
```



