Supervised Sparse Soft-Structured PCA with msma

Overview

Version 4.0 adds opt-in S4PCA for a single X matrix. The default structure.method="none" preserves Version 3.2 behavior.

set.seed(4)
X <- scale(matrix(rnorm(50*12),50,12))
Z <- as.numeric(scale(.7*X[,1]-.4*X[,5]+rnorm(50)))
fit <- msma(X,Z=Z,comp=3,lambdaX=.1,muX=.8,
            structure.method="soft",gammaX=.1,niterS4=30,
            scaling=FALSE,intseed=4)
fit$W
#>             comp1       comp2      comp3
#>  [1,]  0.49522482  0.00000000  0.3741047
#>  [2,]  0.00000000 -0.60676950  0.0000000
#>  [3,]  0.03962165  0.00000000  0.0000000
#>  [4,]  0.00000000  0.76931698  0.0000000
#>  [5,] -0.60036678 -0.02819915  0.0000000
#>  [6,]  0.00000000  0.00000000  0.8124229
#>  [7,]  0.23811293  0.00000000 -0.3525771
#>  [8,]  0.26924703  0.00000000 -0.2001787
#>  [9,]  0.29968506  0.00000000 -0.1887660
#> [10,]  0.00000000  0.00000000  0.0000000
#> [11,]  0.30962252  0.00000000  0.0000000
#> [12,]  0.27905782 -0.19795697  0.0000000
fit$overlap_all
#> [1] 0.5
fit$overlap_selected
#> [1] 0.5454545
fit$diagnostics
#> $engine
#> [1] "reference-compatible"
#> 
#> $iterations_per_component
#> [1] 30 30 30
#> 
#> $extracted_components
#> [1] 3
#> 
#> $requested_components
#> [1] 3
#> 
#> $terminated_early
#> [1] FALSE
#> 
#> $selected_per_component
#> comp1 comp2 comp3 
#>     8     4     5 
#> 
#> $selected_variables
#> [1] 11
#> 
#> $repeated_variables
#> [1] 6
#> 
#> $convergence_rule
#> [1] "fixed_iterations"
#> 
#> $reference_compatible
#> [1] TRUE

Hard exclusion is selected with structure.method="exclusive". Version 4.0 initially limits structured PCA to single-matrix PCA, scalar comp and lambdaX, and vector or one-column Z.

sessionInfo()
#> R version 4.5.2 (2025-10-31 ucrt)
#> Platform: x86_64-w64-mingw32/x64
#> Running under: Windows 11 x64 (build 26300)
#> 
#> Matrix products: default
#>   LAPACK version 3.12.1
#> 
#> locale:
#> [1] LC_COLLATE=C                    LC_CTYPE=Japanese_Japan.utf8   
#> [3] LC_MONETARY=Japanese_Japan.utf8 LC_NUMERIC=C                   
#> [5] LC_TIME=Japanese_Japan.utf8    
#> 
#> time zone: Asia/Tokyo
#> tzcode source: internal
#> 
#> attached base packages:
#> [1] stats     graphics  grDevices utils     datasets  methods   base     
#> 
#> other attached packages:
#> [1] msma_4.0
#> 
#> loaded via a namespace (and not attached):
#>  [1] digest_0.6.39   R6_2.6.1        fastmap_1.2.0   xfun_0.55      
#>  [5] cachem_1.1.0    knitr_1.51      htmltools_0.5.9 rmarkdown_2.31 
#>  [9] lifecycle_1.0.5 cli_3.6.6       sass_0.4.10     jquerylib_0.1.4
#> [13] compiler_4.5.2  tools_4.5.2     evaluate_1.0.5  bslib_0.9.0    
#> [17] yaml_2.3.12     otel_0.2.0      rlang_1.3.0     jsonlite_2.0.0

Repeated split conformal model selection

The candidate grid can be evaluated by repeated calibration splits. The code below uses a deliberately small grid for illustration.

selection <- s4pca_conformal_select(
  X, Z,
  lambdaX = c(0.05, 0.10),
  gammaX = c(0, 0.10),
  comp = 2:3,
  repeats = 5,
  alpha = 0.10,
  max_overlap = 0.50,
  muX = 0.8,
  intseed = 4
)
selection$selected
selection$summary
selection$fit