# Cap OpenMP to 2 threads so forest training in this vignette never uses more # than two cores when the vignette is rebuilt under R CMD check (CRAN policy). Sys.setenv(OMP_NUM_THREADS = "2") knitr::opts_chunk$set( collapse = TRUE, comment = "#>", fig.width = 7, fig.height = 5 )
ppforest2 builds oblique decision trees and random forests using projection pursuit. Instead of splitting on a single variable at each node, it finds a linear combination of variables that best separates the groups.
Train a projection-pursuit tree on the iris dataset:
library(ppforest2) tree <- pptr(Species ~ ., data = iris, seed = 0) tree
The tree splits on linear projections of the features. Use summary() to see variable importance:
summary(tree)
Predict new observations:
preds <- predict(tree, iris) table(predicted = preds, actual = iris$Species)
Train a forest of 100 trees, each considering 2 randomly chosen variables at each split:
forest <- pprf(Species ~ ., data = iris, size = 100, n_vars = 2, seed = 0) summary(forest)
The summary shows the OOB (out-of-bag) error estimate and three variable importance measures:
Predict with class probabilities:
probs <- predict(forest, iris[1:5, ], type = "prob") probs
ppforest2 provides several plot types via plot() (requires ggplot2):
# Tree structure with projected data histograms at each node plot(tree, type = "structure")
# Variable importance bar chart plot(tree, type = "importance")
# Data projected onto the first split's projection vector plot(tree, type = "projection")
# Decision boundaries for two selected variables plot(tree, type = "boundaries")
For forests, variable importance is the only global visualization available. Individual trees can be plotted by specifying a tree_index.
plot(forest)
plot(forest, type = "structure", tree_index = 1)
Set lambda > 0 to use Penalized Discriminant Analysis instead of LDA. This can help when features are highly correlated:
tree_pda <- pptr(Species ~ ., data = iris, lambda = 0.5, seed = 0) summary(tree_pda)
For more control, you can pass strategy objects directly instead of using the shortcut parameters (lambda, n_vars, p_vars). This is equivalent but makes the strategy choice explicit:
# These two calls produce identical results: forest_shortcut <- pprf(Species ~ ., data = iris, size = 10, lambda = 0.5, n_vars = 2, seed = 0) forest_explicit <- pprf(Species ~ ., data = iris, size = 10, pp = pp_pda(0.5), vars = vars_uniform(n_vars = 2), seed = 0) all.equal(predict(forest_shortcut, iris), predict(forest_explicit, iris))
Available strategy constructors:
pp_pda(lambda) — PDA projection pursuit (lambda = 0 for LDA)vars_uniform(n_vars) or vars_uniform(p_vars) — random variable selectionvars_all() — use all variables (default for single trees)cutpoint_mean_of_means() — midpoint split rule (default)Strategy objects can also be passed as engine arguments in parsnip:
library(parsnip) spec <- pp_rand_forest(trees = 10) |> set_engine("ppforest2", pp = pp_pda(0.5), vars = vars_uniform(n_vars = 2)) |> set_mode("classification") fit <- spec |> fit(Species ~ ., data = iris) predict(fit, iris[1:5, ])
ppforest2 integrates with the tidymodels ecosystem via parsnip:
library(parsnip) # Single tree spec <- pp_tree(penalty = 0.5) |> set_engine("ppforest2") |> set_mode("classification") fit <- spec |> fit(Species ~ ., data = iris) predict(fit, iris[1:5, ])
# Random forest spec <- pp_rand_forest(trees = 50, mtry = 2, penalty = 0.5) |> set_engine("ppforest2") |> set_mode("classification") fit <- spec |> fit(Species ~ ., data = iris) predict(fit, iris[1:5, ], type = "prob")
Any scripts or data that you put into this service are public.
Add the following code to your website.
For more information on customizing the embed code, read Embedding Snippets.