## ----setup, include = FALSE---------------------------------------------------
fixture_dir <- "tidymodels"
recording <- nzchar(Sys.getenv("FOUNDRY_RECORD_DOCS"))
have_fixtures <- dir.exists(fixture_dir) && length(list.files(fixture_dir)) > 0
run_api <- requireNamespace("httptest2", quietly = TRUE) &&
  (recording || have_fixtures)
library(foundryR)
if (run_api) {
  httptest2::start_vignette(fixture_dir)
}
knitr::opts_chunk$set(collapse = TRUE, comment = "#>", eval = run_api,
  fig.width = 7, fig.height = 4.5, out.width = "100%")

## ----install, eval = FALSE----------------------------------------------------
# install.packages("tidymodels")

## ----libraries, message = FALSE, eval = requireNamespace("tidymodels", quietly = TRUE)----
library(tidymodels)
library(foundryR)

## ----data, eval = requireNamespace("tidymodels", quietly = TRUE)--------------
reviews <- tibble(
  text = c(
    "The examples made the concepts easy to apply.",
    "The assignment instructions were clear and useful.",
    "The instructor explained the hard parts carefully.",
    "The labs helped me practice each method.",
    "The readings connected well to the lectures.",
    "The feedback on drafts helped me improve.",
    "The setup steps were confusing and slow.",
    "The grading rubric was hard to interpret.",
    "The software instructions skipped important details.",
    "The lectures moved too quickly for me.",
    "The final project needed more guidance.",
    "The examples did not match the homework."
  ),
  sentiment = factor(rep(c("positive", "negative"), each = 6))
)

## ----basic-recipe, eval = requireNamespace("tidymodels", quietly = TRUE)------
recipe_spec <- recipe(sentiment ~ text, data = reviews) |>
  step_foundry_embed(
    text,
    model = "text-embedding-3-small",
    keep_original = FALSE
  )

recipe_spec

## ----prep-bake, eval = run_api && requireNamespace("tidymodels", quietly = TRUE)----
prepped_recipe <- prep(recipe_spec, training = reviews)
tidy(prepped_recipe, number = 1)

baked_data <- bake(prepped_recipe, new_data = NULL)
baked_data[, c("sentiment", "emb_text_1", "emb_text_2", "emb_text_3")]

## ----pca-workflow, eval = run_api && requireNamespace("tidymodels", quietly = TRUE)----
set.seed(42)
split <- initial_split(reviews, prop = 0.75, strata = sentiment)
train_data <- training(split)
test_data <- testing(split)

embedding_recipe <- recipe(sentiment ~ text, data = train_data) |>
  step_foundry_embed(
    text,
    model = "text-embedding-3-small",
    keep_original = FALSE,
    cache = "disk"
  ) |>
  step_normalize(all_numeric_predictors()) |>
  step_pca(all_numeric_predictors(), num_comp = 3)

log_reg_spec <- logistic_reg() |>
  set_engine("glm") |>
  set_mode("classification")

sentiment_workflow <- workflow() |>
  add_recipe(embedding_recipe) |>
  add_model(log_reg_spec)

fitted_workflow <- fit(sentiment_workflow, data = train_data)

predictions <- predict(fitted_workflow, test_data) |>
  bind_cols(test_data["sentiment"])

predictions

## ----prediction-summary, echo = FALSE, results = "asis"-----------------------
correct <- sum(as.character(predictions$.pred_class) == as.character(predictions$sentiment))
cat(sprintf(
  "The workflow labels %d of the %d held-out comments correctly, but a test set of %d rows says almost nothing about accuracy. With real data, use enough labeled rows for the model you fit, or a penalized model such as `glmnet`, and estimate accuracy by resampling as shown below.\n",
  correct, nrow(predictions), nrow(predictions)
))

## ----dimensions, eval = requireNamespace("tidymodels", quietly = TRUE)--------
compact_recipe <- recipe(sentiment ~ text, data = reviews) |>
  step_foundry_embed(
    text,
    model = "text-embedding-3-small",
    dimensions = 256,
    keep_original = FALSE
  )

## ----multi-column, eval = requireNamespace("tidymodels", quietly = TRUE)------
ticket_data <- tibble(
  subject = c("Login failure", "Billing question"),
  body = c("Password reset link expired.", "Invoice total looks too high."),
  escalated = factor(c("yes", "no"))
)

ticket_recipe <- recipe(escalated ~ ., data = ticket_data) |>
  step_foundry_embed(subject, model = "text-embedding-3-small",
                     prefix = "subject_") |>
  step_foundry_embed(body, model = "text-embedding-3-small",
                     prefix = "body_")

## ----precompute, eval = run_api && requireNamespace("tidymodels", quietly = TRUE)----
embedded_reviews <- foundry_embed_batch(
  reviews$text,
  model = "text-embedding-3-small",
  batch_size = 6,
  max_active = 2
)

embedding_matrix <- do.call(rbind, embedded_reviews$embedding)
embedding_cols <- tibble::as_tibble(
  embedding_matrix,
  .name_repair = function(x) paste0("emb_", seq_along(x))
)

precomputed <- bind_cols(
  reviews["sentiment"],
  embedding_cols
)

precomputed[, c("sentiment", "emb_1", "emb_2", "emb_3")]

## ----reproducibility-summary, echo = FALSE, results = "asis"------------------
recipe_emb <- as.matrix(baked_data[, grep("^emb_text_", names(baked_data))])
batch_emb <- as.matrix(precomputed[, grep("^emb_", names(precomputed))])
max_diff <- max(abs(recipe_emb - batch_emb))
cosine <- rowSums(recipe_emb * batch_emb) /
  (sqrt(rowSums(recipe_emb^2)) * sqrt(rowSums(batch_emb^2)))
if (max_diff == 0) {
  cat("The recipe and `foundry_embed_batch()` returned identical vectors for the same comments in this recording.\n")
} else {
  cat(sprintf(
    "The recipe and `foundry_embed_batch()` embedded the same %d comments in separate calls, and the vectors are not identical. The largest difference in any coordinate is %s, though every pair has a cosine similarity of at least %s. Store the embeddings you analyze, so that a rerun uses the same numbers.\n",
    nrow(batch_emb), format(signif(max_diff, 2)), format(floor(min(cosine) * 1e4) / 1e4, nsmall = 4)
  ))
}

## ----cv, eval = FALSE---------------------------------------------------------
# folds <- vfold_cv(train_data, v = 5, strata = sentiment)
# 
# cv_results <- fit_resamples(
#   sentiment_workflow,
#   resamples = folds,
#   metrics = metric_set(accuracy, roc_auc)
# )
# 
# collect_metrics(cv_results)

## ----rate-limit, eval = FALSE-------------------------------------------------
# small_sample <- reviews |> slice_sample(n = 100)
# prepped <- prep(recipe_spec, training = small_sample)

## ----credentials, eval = FALSE------------------------------------------------
# foundry_check_setup()
# foundry_set_endpoint(Sys.getenv("AZURE_FOUNDRY_ENDPOINT"))
# foundry_set_key("your-api-key")

## ----cleanup, include = FALSE, eval = TRUE------------------------------------
if (run_api) {
  httptest2::end_vignette()
}

