Nothing
## ----setup, include = FALSE---------------------------------------------------
fixture_dir <- "tidymodels"
recording <- nzchar(Sys.getenv("FOUNDRY_RECORD_DOCS"))
have_fixtures <- dir.exists(fixture_dir) && length(list.files(fixture_dir)) > 0
run_api <- requireNamespace("httptest2", quietly = TRUE) &&
(recording || have_fixtures)
have_tidymodels <- requireNamespace("tidymodels", quietly = TRUE)
# Attach foundryR before start_vignette(): httptest2 only sources the package's
# inst/httptest2/start-vignette.R (which sets replay placeholders) from attached
# packages.
library(foundryR)
if (run_api) {
httptest2::start_vignette(fixture_dir)
}
knitr::opts_chunk$set(
collapse = TRUE,
comment = "#>",
eval = run_api && have_tidymodels
)
## ----install, eval = FALSE----------------------------------------------------
# install.packages("tidymodels")
## ----basic-recipe, eval = have_tidymodels-------------------------------------
library(tidymodels)
library(foundryR)
reviews <- tibble(
text = c(
"This product is useful and easy to use.",
"The setup was confusing and slow.",
"The examples were clear and helpful.",
"I needed better instructions."
),
sentiment = factor(c("positive", "negative", "positive", "negative"))
)
recipe_spec <- recipe(sentiment ~ text, data = reviews) %>%
step_foundry_embed(
text,
model = "text-embedding-3-small",
keep_original = FALSE
)
recipe_spec
## ----prep-bake----------------------------------------------------------------
prepped_recipe <- prep(recipe_spec, training = reviews)
baked_data <- bake(prepped_recipe, new_data = NULL)
baked_data
## ----full-pipeline, eval = FALSE----------------------------------------------
# library(tidymodels)
# library(foundryR)
#
# # Load your data
# set.seed(123)
# reviews <- tibble(
# review_text = c(
# # Positive reviews
# "Absolutely love this product! Works perfectly.",
# "Great quality and fast shipping. Very satisfied.",
# "Best purchase I've made this year. Highly recommend!",
# "Exceeded all expectations. Will buy again.",
# "Perfect fit and great value for money.",
# # Negative reviews
# "Complete waste of money. Broke after one use.",
# "Terrible customer service. Never buying again.",
# "Poor quality, doesn't work as advertised.",
# "Disappointed. Much smaller than expected.",
# "Arrived damaged and took forever to ship."
# ),
# sentiment = factor(rep(c("positive", "negative"), each = 5))
# )
#
# # Split data
# splits <- initial_split(reviews, prop = 0.8, strata = sentiment)
# train_data <- training(splits)
# test_data <- testing(splits)
#
# # Define recipe with embeddings
# embedding_recipe <- recipe(sentiment ~ review_text, data = train_data) %>%
# step_foundry_embed(
# review_text,
# model = "text-embedding-3-small",
# keep_original = FALSE
# ) %>%
# step_normalize(all_numeric_predictors()) # Normalize embedding dimensions
#
# # Define model
# log_reg_spec <- logistic_reg() %>%
# set_engine("glm") %>%
# set_mode("classification")
#
# # Create workflow
# sentiment_workflow <- workflow() %>%
# add_recipe(embedding_recipe) %>%
# add_model(log_reg_spec)
#
# # Fit the model
# fitted_workflow <- fit(sentiment_workflow, data = train_data)
#
# # Make predictions on test data
# predictions <- predict(fitted_workflow, test_data) %>%
# bind_cols(test_data)
#
# # Evaluate
# predictions %>%
# metrics(truth = sentiment, estimate = .pred_class)
## ----dimensions, eval = have_tidymodels---------------------------------------
recipe_spec <- recipe(sentiment ~ text, data = reviews) %>%
step_foundry_embed(
text,
model = "text-embedding-3-small",
dimensions = 256, # Reduce from 1536 to 256
keep_original = FALSE
)
## ----multi-column, eval = have_tidymodels-------------------------------------
# Data with multiple text fields
data <- tibble(
title = c("Great Product", "Terrible Experience"),
description = c("Works as expected", "Broke immediately"),
outcome = c(1, 0)
)
recipe_spec <- recipe(outcome ~ ., data = data) %>%
step_foundry_embed(title, model = "text-embedding-3-small",
prefix = "title_") %>%
step_foundry_embed(description, model = "text-embedding-3-small",
prefix = "desc_") %>%
step_rm(title, description) # Remove original text columns
## ----keep-original, eval = have_tidymodels------------------------------------
recipe_spec <- recipe(sentiment ~ text, data = reviews) %>%
step_foundry_embed(
text,
model = "text-embedding-3-small",
keep_original = TRUE # Keep the text column
)
# Useful when you also want to apply other text processing
## ----prefix, eval = have_tidymodels-------------------------------------------
recipe_spec <- recipe(sentiment ~ text, data = reviews) %>%
step_foundry_embed(
text,
model = "text-embedding-3-small",
prefix = "embed_" # Columns will be embed_001, embed_002, etc.
)
## ----precompute---------------------------------------------------------------
embedded_reviews <- foundry_embed(
reviews$text,
model = "text-embedding-3-small"
)
embedding_matrix <- do.call(rbind, embedded_reviews$embedding)
embedding_cols <- as_tibble(embedding_matrix, .name_repair = "unique")
precomputed <- bind_cols(
reviews["sentiment"],
embedding_cols
)
precomputed[, 1:4]
## ----cv, eval = FALSE---------------------------------------------------------
# # Create CV folds
# folds <- vfold_cv(train_data, v = 5, strata = sentiment)
#
# # Fit resamples
# cv_results <- fit_resamples(
# sentiment_workflow,
# resamples = folds,
# metrics = metric_set(accuracy, roc_auc)
# )
#
# # Collect metrics
# collect_metrics(cv_results)
## ----tuning, eval = FALSE-----------------------------------------------------
# # Recipe with tunable dimensions
# tunable_recipe <- recipe(sentiment ~ text, data = train_data) %>%
# step_foundry_embed(
# text,
# model = "text-embedding-3-small",
# dimensions = tune(), # Will be tuned
# keep_original = FALSE
# ) %>%
# step_normalize(all_numeric_predictors())
#
# # Model with tunable parameters
# rf_spec <- rand_forest(
# mtry = tune(),
# trees = 500,
# min_n = tune()
# ) %>%
# set_engine("ranger") %>%
# set_mode("classification")
#
# # Workflow
# tunable_workflow <- workflow() %>%
# add_recipe(tunable_recipe) %>%
# add_model(rf_spec)
#
# # Define grid
# grid <- grid_regular(
# dimensions(range = c(128, 512)), # Embedding dimensions
# mtry(range = c(10, 50)),
# min_n(range = c(2, 10)),
# levels = 3
# )
#
# # Tune only after estimating the API calls and cost.
# tune_results <- tune_grid(
# tunable_workflow,
# resamples = folds,
# grid = grid,
# metrics = metric_set(accuracy, roc_auc)
# )
#
# # Best parameters
# show_best(tune_results, metric = "roc_auc")
## ----memory, eval = TRUE------------------------------------------------------
# Estimate memory for 10,000 texts
n_texts <- 10000
n_dims <- 1536
bytes_per_double <- 8
memory_mb <- (n_texts * n_dims * bytes_per_double) / 1024^2
print(paste(round(memory_mb), "MB for embeddings alone"))
## ----troubleshoot, eval = have_tidymodels-------------------------------------
# Use a unique prefix if reusing recipes
recipe_spec <- recipe(sentiment ~ text, data = reviews) %>%
step_foundry_embed(text, model = "my-model",
prefix = paste0("v", format(Sys.time(), "%H%M%S"), "_"))
## ----rate-limit, eval = FALSE-------------------------------------------------
# # Prepare in smaller batches
# small_sample <- reviews %>% slice_sample(n = 100)
# prepped <- prep(recipe_spec, training = small_sample)
## ----credentials, eval = FALSE------------------------------------------------
# # Check setup
# foundry_check_setup()
#
# # Set credentials if needed.
# foundry_set_endpoint(Sys.getenv("AZURE_FOUNDRY_ENDPOINT"))
# foundry_set_key("your-api-key")
## ----cleanup, include = FALSE, eval = TRUE------------------------------------
if (run_api) {
httptest2::end_vignette()
}
Any scripts or data that you put into this service are public.
Add the following code to your website.
For more information on customizing the embed code, read Embedding Snippets.