## ----setup, include = FALSE---------------------------------------------------
fixture_dir <- "audio"
recording <- nzchar(Sys.getenv("FOUNDRY_RECORD_DOCS"))
have_fixtures <- dir.exists(fixture_dir) && length(list.files(fixture_dir)) > 0
run_api <- requireNamespace("httptest2", quietly = TRUE) &&
  (recording || have_fixtures)
library(foundryR)
if (run_api) {
  httptest2::start_vignette(fixture_dir)
}
knitr::opts_chunk$set(collapse = TRUE, comment = "#>", eval = run_api,
  fig.width = 7, fig.height = 4.5, out.width = "100%")

embed_audio <- function(path) {
  if (!requireNamespace("base64enc", quietly = TRUE)) {
    return(invisible(NULL))
  }
  uri <- paste0("data:audio/mpeg;base64,", base64enc::base64encode(path))
  knitr::asis_output(
    sprintf(
      '<audio controls src="%s">Your browser does not support audio playback.</audio>',
      uri
    )
  )
}

## ----library, eval = TRUE, message = FALSE------------------------------------
library(foundryR)

## ----sample, eval = TRUE------------------------------------------------------
sample_audio <- system.file("extdata", "samples", "jfk.wav", package = "foundryR")
basename(sample_audio)

## ----transcribe---------------------------------------------------------------
transcript <- foundry_transcribe(sample_audio)

transcript$text
transcript[, c("model", "duration_ms", "language")]

## ----transcribe-phrases-------------------------------------------------------
head(transcript$phrases[[1]][, c("text", "offset_ms", "duration_ms")])

## ----mai-transcribe, eval = FALSE---------------------------------------------
# foundry_set_speech_endpoint(Sys.getenv("AZURE_FOUNDRY_SPEECH_ENDPOINT"))
# foundry_set_speech_key("your-speech-key")
# 
# foundry_transcribe(
#   sample_audio,
#   model = "mai-transcribe-2"
# )

## ----translate-synth----------------------------------------------------------
spanish_path <- tempfile(fileext = ".mp3")
spanish_clip <- foundry_speak(
  paste(
    "Buenos dias a todos. En la reunion de hoy revisamos la encuesta a los docentes.",
    "La mayoria pidio mas tiempo para preparar las clases y menos tareas administrativas.",
    "El equipo presentara un plan el proximo mes."
  ),
  model = "gpt-4o-mini-tts",
  voice = "verse",
  path = spanish_path
)

## ----translate-play, echo = FALSE---------------------------------------------
embed_audio(spanish_clip$path)

## ----spanish-transcribe-------------------------------------------------------
spanish <- foundry_transcribe(spanish_clip$path, locales = "es-ES")
spanish$text

english <- foundry_response(
  spanish$text,
  instructions = "Translate the text into English. Return only the translation."
)
english$output_text

## ----translate----------------------------------------------------------------
whisper <- foundry_translate_audio(
  spanish_clip$path,
  service = "openai",
  model = "whisper",
  api = "deployment"
)

whisper$text

## ----translate-summary, echo = FALSE, results = "asis"------------------------
words <- function(x) {
  x <- tolower(iconv(x, "UTF-8", "ASCII//TRANSLIT"))
  unique(strsplit(gsub("[^a-z ]", " ", x), "\\s+")[[1]])
}
source_words <- setdiff(words(spanish$text), "")
shared <- mean(source_words %in% words(whisper$text))
if (shared > 0.6) {
  cat(sprintf(
    "The Whisper translation route returned the clip in Spanish: %.0f%% of the words in the Spanish transcript appear in its output. Whisper translation should return English, so check the language of every output before you use this route in a pipeline.\n",
    100 * shared
  ))
} else {
  cat("Whisper returned English text for this clip. Check the language of every output before you use this route in a pipeline, because Whisper can return untranslated text.\n")
}

## ----code-transcript----------------------------------------------------------
transcript_schema <- foundry_schema(
  topic = schema_enum(
    c("workload", "curriculum", "facilities", "other"),
    "Main topic of the passage."
  ),
  tone = schema_enum(
    c("formal", "informal", "urgent", "reflective"),
    "Overall tone of the speaker."
  )
)

meeting_code <- foundry_extract(
  spanish$text,
  schema = transcript_schema,
  schema_name = "TranscriptCode"
)

meeting_code[, c("topic", "tone", ".status", ".error")]

## ----code-speech--------------------------------------------------------------
speech_code <- foundry_extract(
  transcript$text,
  schema = transcript_schema,
  schema_name = "TranscriptCode"
)

speech_code[, c(".status", ".error")]
speech_code$.error_msg

## ----code-speech-summary, echo = FALSE, results = "asis"----------------------
if (isTRUE(speech_code$.error) && grepl("content_filter", speech_code$.error_msg)) {
  cat("In this recording the content filter stopped the response, although the request asks only for two labels. Widely quoted text, such as a famous speech, can trip the filter. If you code texts like these, ask your Azure administrator about the content filter configuration on the deployment, and report how many rows failed next to any coded shares.\n")
} else {
  cat("In this recording the JFK transcript was coded without an error. Deployments with stricter content filters can still stop responses about widely quoted text, so count failed rows before you report coded shares.\n")
}

## ----speak--------------------------------------------------------------------
speech_path <- tempfile(fileext = ".mp3")
speech <- foundry_speak(
  "Please read each survey question before choosing an answer.",
  model = "gpt-4o-mini-tts",
  voice = "verse",
  path = speech_path
)

speech[, c("bytes", "model", "voice", "format")]

## ----speak-play, echo = FALSE-------------------------------------------------
embed_audio(speech$path)

## ----cleanup, include = FALSE, eval = TRUE------------------------------------
if (run_api) {
  unlink(c(speech_path, spanish_path))
  httptest2::end_vignette()
}

