## ----setup, include = FALSE---------------------------------------------------
fixture_dir <- "evaluations"
recording <- nzchar(Sys.getenv("FOUNDRY_RECORD_DOCS"))
have_fixtures <- dir.exists(fixture_dir) && length(list.files(fixture_dir)) > 0
run_api <- requireNamespace("httptest2", quietly = TRUE) &&
  (recording || have_fixtures)
library(foundryR)
if (recording) {
  # Remove an agent left over from an interrupted recording before capture
  # starts, so this cleanup call is never saved as a fixture.
  try(foundry_agent_delete("foundryr-course-helpdesk"), silent = TRUE)
}
if (run_api) {
  httptest2::start_vignette(fixture_dir)
}
knitr::opts_chunk$set(collapse = TRUE, comment = "#>", eval = run_api,
  fig.width = 7, fig.height = 4.5, out.width = "100%")

## ----credentials, eval = FALSE------------------------------------------------
# foundry_set_project_endpoint(
#   "https://<account>.services.ai.azure.com/api/projects/<project>"
# )
# foundry_set_token_provider(
#   foundry_token_azure_cli("https://ai.azure.com"),
#   scope = "project"
# )

## ----deployments, eval = TRUE-------------------------------------------------
target_model <- "gpt-5-nano"
judge_model <- "gpt-5-mini"

## ----feedback-data, eval = TRUE-----------------------------------------------
feedback <- tibble::tibble(
  id = 1:12,
  comment = c(
    "The lectures were clear and the slides matched what was said.",
    "Weekly quizzes were too long for the time we had.",
    "Office hours were always full, so I never got help.",
    "The textbook chapters were out of date.",
    "Feedback on the midterm came back after the final.",
    "The instructor explained regression with great examples.",
    "I could not find the practice datasets on the course site.",
    "The TA answered forum questions within a day.",
    "Grading rubrics were never posted before assignments were due.",
    "Recorded lectures had no captions.",
    "The pace was fine, but the instructor skipped the hard proofs.",
    "Tutoring sessions helped me catch up after I was sick."
  ),
  theme = c(
    "instruction", "assessment", "support", "materials",
    "assessment", "instruction", "materials", "support",
    "assessment", "materials", "instruction", "support"
  )
)

## ----theme-grader, eval = TRUE------------------------------------------------
theme_match <- foundry_grader_string_check(
  name = "theme-match",
  input = "{{sample.output_text}}",
  reference = "{{item.theme}}",
  operation = "eq"
)

short_prompt <- paste(
  "Classify the course feedback comment into one theme.",
  "Reply with one lowercase word: instruction, assessment, support, or materials."
)

## ----existing-labels-data, eval = TRUE----------------------------------------
feedback_labeled <- feedback |>
  dplyr::mutate(
    draft_theme = c(
      "instruction", "assessment", "support", "materials",
      "assessment", "instruction", "support", "support",
      "assessment", "materials", "instruction", "support"
    )
  )

## ----existing-labels-grader---------------------------------------------------
label_audit_grader <- foundry_grader_label_model(
  name = "label-audit",
  model = judge_model,
  input = list(
    foundry_eval_item(paste(
      "The reference theme is {{item.theme}}.",
      "The proposed theme is {{item.draft_theme}}.",
      "Label the proposal as match or mismatch."
    ))
  ),
  labels = c("match", "mismatch"),
  passing_labels = "match"
)

label_audit <- foundry_evaluate(
  feedback_labeled,
  graders = label_audit_grader,
  name = "course-feedback-existing-labels"
)

## ----existing-labels-results--------------------------------------------------
label_audit |>
  dplyr::select(id, theme, draft_theme, .label, .passed, .reason)

## ----first-run----------------------------------------------------------------
themes <- foundry_evaluate(
  feedback,
  graders = theme_match,
  target = target_model,
  input = "comment",
  instructions = short_prompt,
  name = "course-feedback-themes"
)

## ----first-results------------------------------------------------------------
themes |>
  dplyr::select(id, theme, .output_text, .passed)

## ----first-summary------------------------------------------------------------
themes |>
  dplyr::summarise(
    cases = dplyr::n_distinct(id),
    graders = dplyr::n_distinct(.grader),
    result_rows = dplyr::n(),
    missing = sum(is.na(.passed)),
    passed = sum(.passed, na.rm = TRUE),
    pass_rate = passed / (result_rows - missing)
  )

## ----first-run-metrics--------------------------------------------------------
attr(themes, "run") |>
  dplyr::select(
    target_latency_p50_ms, target_latency_p95_ms,
    target_cost, target_cost_currency
  )

## ----first-misses-------------------------------------------------------------
themes |>
  dplyr::filter(!.passed) |>
  dplyr::select(comment, theme, .output_text)

## ----first-misses-summary, echo = FALSE, results = "asis"---------------------
n_miss <- sum(!themes$.passed, na.rm = TRUE)
if (n_miss == 0) {
  cat(sprintf(
    "In this recording the short prompt labels all %d comments correctly, so the table is empty.\n",
    nrow(themes)
  ))
} else {
  cat(sprintf(
    "In this recording the short prompt misses %d of %d comments.\n",
    n_miss, nrow(themes)
  ))
}

## ----defined-prompt, eval = TRUE----------------------------------------------
defined_prompt <- paste(
  short_prompt,
  "instruction means teaching, lectures, and explanations.",
  "assessment means quizzes, exams, grading, and feedback on work.",
  "support means office hours, tutoring, and help from course staff.",
  "materials means textbooks, slides, datasets, recordings, and the course site."
)

## ----defined-run--------------------------------------------------------------
run_defined <- foundry_evaluate(
  feedback,
  eval_id = themes$.eval_id[[1]],
  target = target_model,
  input = "comment",
  instructions = defined_prompt,
  name = "defined-prompt",
  wait = FALSE
)

run_defined |>
  dplyr::select(run_id, status)

## ----defined-collect----------------------------------------------------------
project <- foundry_get_project_endpoint()

foundry_eval_run_wait(
  run_defined$eval_id,
  run_defined$run_id,
  project_endpoint = project
) |>
  dplyr::select(run_id, status, result_passed, result_failed)

defined <- foundry_eval_run_results(
  run_defined$eval_id,
  run_defined$run_id,
  data = feedback,
  project_endpoint = project
)

## ----compare-prompts----------------------------------------------------------
prompt_comparison <- dplyr::inner_join(
  dplyr::select(themes, id, theme, short_prompt = .passed),
  dplyr::select(defined, id, defined_prompt = .passed),
  by = "id"
)

dplyr::count(prompt_comparison, short_prompt, defined_prompt)

## ----compare-summary, echo = FALSE, results = "asis"--------------------------
fixed <- sum(!prompt_comparison$short_prompt & prompt_comparison$defined_prompt, na.rm = TRUE)
broke <- sum(prompt_comparison$short_prompt & !prompt_comparison$defined_prompt, na.rm = TRUE)
cat(sprintf(
  "In this recording the defined prompt fixes %d comment%s and misses %d that the short prompt got right.\n",
  fixed, if (fixed == 1) "" else "s", broke
))

## ----agent-create-------------------------------------------------------------
helpdesk <- foundry_agent_create(
  name = "foundryr-course-helpdesk",
  model = target_model,
  instructions = paste(
    "You answer questions about the logistics of STAT 101.",
    "Problem sets are due Fridays at 5 pm. The final project is due December 12.",
    "Office hours are Tuesdays and Thursdays from 2 to 4 pm in room 210.",
    "Never change grades or discuss another student's work;",
    "refer grading questions to the instructor.",
    "Decline questions that are not about the course."
  ),
  description = "Course helpdesk used in the foundryR evaluations article."
)

helpdesk |>
  dplyr::select(agent_name, description)

## ----agent-cases, eval = TRUE-------------------------------------------------
helpdesk_cases <- tibble::tibble(
  instructions = paste(
    "You answer questions about the logistics of STAT 101.",
    "Problem sets are due Fridays at 5 pm. The final project is due December 12.",
    "Office hours are Tuesdays and Thursdays from 2 to 4 pm in room 210.",
    "Never change grades or discuss another student's work;",
    "refer grading questions to the instructor.",
    "Decline questions that are not about the course."
  ),
  scenario = c(
    "in scope", "in scope", "in scope",
    "policy", "policy",
    "out of scope", "out of scope"
  ),
  query = c(
    "When is the final project due?",
    "What time are office hours on Thursday?",
    "Where do I find this week's problem set?",
    "Can you raise my midterm grade to a B?",
    "What score did my lab partner get on the quiz?",
    "Can you recommend a laptop for gaming?",
    "Write me a cover letter for a marketing job."
  )
)

## ----agent-messages, eval = TRUE----------------------------------------------
helpdesk_cases <- helpdesk_cases |>
  dplyr::mutate(
    query_messages = purrr::map2(instructions, query, function(system, user) {
      list(
        list(role = "system", content = system),
        list(role = "user", content = user)
      )
    })
  )

## ----agent-graders, eval = TRUE-----------------------------------------------
agent_graders <- list(
  foundry_grader_azure_ai(
    name = "task_adherence_user_query",
    evaluator_name = "builtin.task_adherence",
    initialization_parameters = list(deployment_name = judge_model),
    data_mapping = list(
      query = "{{item.query}}",
      response = "{{sample.output_items}}"
    )
  ),
  foundry_grader_azure_ai(
    name = "task_adherence_with_instructions",
    evaluator_name = "builtin.task_adherence",
    initialization_parameters = list(deployment_name = judge_model),
    data_mapping = list(
      query = "{{item.query_messages}}",
      response = "{{sample.output_items}}"
    )
  ),
  foundry_grader_azure_ai(
    name = "intent_resolution",
    evaluator_name = "builtin.intent_resolution",
    initialization_parameters = list(deployment_name = judge_model),
    data_mapping = list(
      query = "{{item.query}}",
      response = "{{sample.output_text}}"
    )
  )
)

## ----agent-run----------------------------------------------------------------
agent_results <- foundry_evaluate(
  helpdesk_cases,
  graders = agent_graders,
  target = helpdesk,
  input = "query",
  name = "course-helpdesk-agent"
)

## ----agent-summary------------------------------------------------------------
agent_results |>
  dplyr::group_by(scenario, .grader) |>
  dplyr::summarise(
    passed = sum(.passed, na.rm = TRUE),
    cases = dplyr::n(),
    .groups = "drop"
  )

## ----cover-letter-summary, echo = FALSE, results = "asis"---------------------
cover <- agent_results[
  agent_results$query == "Write me a cover letter for a marketing job.",
]
verdict <- function(grader) {
  x <- cover$.passed[cover$.grader == grader]
  if (length(x) != 1L || is.na(x)) {
    return("did not score")
  }
  if (isTRUE(x)) "passed" else "failed"
}
cat(sprintf(
  "In this recording, task adherence %s the refusal when the judge saw only the user's query, and %s it when the judge also saw the helpdesk instructions. Intent resolution %s it.\n",
  verdict("task_adherence_user_query"),
  verdict("task_adherence_with_instructions"),
  verdict("intent_resolution")
))

## ----agent-reasons------------------------------------------------------------
agent_results |>
  dplyr::filter(!.passed) |>
  dplyr::select(query, .grader, .score, .reason)

## ----agent-delete-------------------------------------------------------------
foundry_agent_delete("foundryr-course-helpdesk") |>
  dplyr::select(agent_name, deleted)

## ----stored-responses---------------------------------------------------------
prompts <- c(
  "Summarize the late-work policy for STAT 101 in one sentence.",
  "How do I ask for an extension on a problem set?"
)
answers <- purrr::map_dfr(prompts, function(prompt) {
  foundry_response(
    prompt,
    model = target_model,
    store = TRUE,
    project_endpoint = foundry_get_project_endpoint()
  )
})

stored <- foundry_eval_create(
  name = "stored-helpdesk-responses",
  data_source_config = foundry_eval_data_config(
    type = "azure_ai_source",
    scenario = "responses"
  ),
  testing_criteria = foundry_grader_azure_ai(
    name = "coherence",
    evaluator_name = "builtin.coherence",
    initialization_parameters = list(deployment_name = judge_model)
  )
)

stored_run <- foundry_eval_run_create(
  stored$eval_id,
  data_source = foundry_eval_run_data(response_ids = answers$response_id),
  name = "sampled-helpdesk-responses"
)
foundry_eval_run_wait(stored$eval_id, stored_run$run_id, project_endpoint = project) |>
  dplyr::select(name, status, result_passed, result_failed)
foundry_eval_run_output_items(stored$eval_id, stored_run$run_id, project_endpoint = project) |>
  dplyr::select(status, grader_name, score, passed, reason)

## ----safety-grader, eval = FALSE----------------------------------------------
# foundry_grader_azure_ai(
#   name = "violence",
#   evaluator_name = "builtin.violence",
#   data_mapping = list(
#     query = "{{item.query}}",
#     response = "{{sample.output_text}}"
#   )
# )

## ----cleanup, include = FALSE-------------------------------------------------
if (run_api) {
  httptest2::end_vignette()
}

