Skip to contents

This recipe extracts metadata from an R package’s source in three steps, each a separate agent run whose typed output feeds the next:

  1. Extract: read DESCRIPTION and R/, and return the package’s metadata.
  2. Categorise: tag the package and summarise each exported function.
  3. Report: write a short summary for people.

Splitting the work lets you check each intermediate result before paying for the next step, and keeps each prompt small and specific. For open-ended exploration where the next step depends on what was found, see the data analysis agent.

Step 1: extract

Describe the output with ellmer types. An array of objects comes back as a data frame:

library(deputy)

metadata_type <- ellmer::type_object(
  name = ellmer::type_string(),
  title = ellmer::type_string(),
  version = ellmer::type_string(),
  authors = ellmer::type_array(
    ellmer::type_object(
      name = ellmer::type_string(),
      role = ellmer::type_string()
    )
  ),
  dependencies = ellmer::type_array(ellmer::type_string()),
  exported_functions = ellmer::type_array(
    ellmer::type_object(
      name = ellmer::type_string(),
      file = ellmer::type_string()
    )
  )
)

The extractor can read files but not change them:

extractor <- Agent$new(
  chat = ellmer::chat("anthropic/claude-sonnet-5"),
  tools = tools_preset("minimal"),
  permissions = permissions_readonly(),
  system_prompt = "You extract package metadata. Read DESCRIPTION, and find
    exported functions by looking for @export tags in R/."
)

step1 <- extractor$run_sync(
  "Extract the metadata of the R package in the working directory.",
  type = metadata_type
)
if (!result_is_success(step1)) {
  cli::cli_abort("Extraction stopped early: {step1$stop_reason}.")
}

metadata <- step1$structured_output
metadata$version
metadata$exported_functions

Before moving on, this is the place to check the extraction, for example that the version matches DESCRIPTION or that no exported function is missing.

Step 2: categorise

The second agent gets the first step’s output as JSON in its task, and may read source files to understand what each function does:

summary_type <- ellmer::type_object(
  categories = ellmer::type_array(
    ellmer::type_enum(c(
      "data", "modeling", "visualization", "infrastructure",
      "testing", "io", "web", "cli"
    ))
  ),
  complexity = ellmer::type_enum(c("simple", "moderate", "complex")),
  functions = ellmer::type_array(
    ellmer::type_object(
      name = ellmer::type_string(),
      summary = ellmer::type_string("One sentence on what the function does.")
    )
  )
)

categoriser <- Agent$new(
  chat = ellmer::chat("anthropic/claude-sonnet-5"),
  tools = tools_preset("minimal"),
  permissions = permissions_readonly(),
  system_prompt = "You categorise R packages and summarise their functions.
    Read source files when a function's name isn't enough."
)

step2 <- categoriser$run_sync(
  paste(
    "Categorise this package and summarise each exported function.",
    jsonlite::toJSON(metadata, auto_unbox = TRUE, pretty = TRUE),
    sep = "\n\n"
  ),
  type = summary_type
)
if (!result_is_success(step2)) {
  cli::cli_abort("Categorisation stopped early: {step2$stop_reason}.")
}

summaries <- step2$structured_output
summaries$complexity
summaries$functions

Step 3: report

The last step needs no tools and returns text:

reporter <- Agent$new(
  chat = ellmer::chat("anthropic/claude-sonnet-5"),
  system_prompt = "You write short, plain package summaries in Markdown, with
    sections for overview, key functions and dependencies."
)

step3 <- reporter$run_sync(paste(
  "Write a summary of this package.",
  jsonlite::toJSON(
    list(metadata = metadata, summaries = summaries),
    auto_unbox = TRUE,
    pretty = TRUE
  ),
  sep = "\n\n"
))

cat(step3$response)

Put it in a function

summarise_package <- function(package_dir = ".") {
  run_step <- function(system_prompt, task, type = NULL, tools = list()) {
    agent <- Agent$new(
      chat = ellmer::chat("anthropic/claude-sonnet-5"),
      tools = tools,
      permissions = permissions_readonly(),
      working_dir = package_dir,
      system_prompt = system_prompt
    )
    result <- agent$run_sync(task, type = type)
    if (!result_is_success(result)) {
      cli::cli_abort("A step stopped early: {result$stop_reason}.")
    }
    result
  }
  as_json <- function(x) jsonlite::toJSON(x, auto_unbox = TRUE)

  step1 <- run_step(
    "You extract package metadata from DESCRIPTION and R/.",
    "Extract the metadata of the R package in the working directory.",
    type = metadata_type,
    tools = tools_preset("minimal")
  )
  step2 <- run_step(
    "You categorise R packages and summarise their functions.",
    paste("Categorise this package:", as_json(step1$structured_output)),
    type = summary_type,
    tools = tools_preset("minimal")
  )
  step3 <- run_step(
    "You write short, plain package summaries in Markdown.",
    paste(
      "Summarise this package:",
      as_json(list(step1$structured_output, step2$structured_output))
    )
  )

  costs <- c(step1$usage$cost_usd, step2$usage$cost_usd, step3$usage$cost_usd)
  list(
    metadata = step1$structured_output,
    summaries = step2$structured_output,
    report = step3$response,
    cost_usd = sum(costs)
  )
}

pipeline <- summarise_package(".")
cat(pipeline$report)

sum() returns NA if any step’s cost is unknown, which is what you want: an incomplete total shouldn’t pass for a real one.

When a step fails

A step can stop early because it hit a limit, or return a value that has the right shape but the wrong content. result_is_success() catches the first. For the second, pass a validate function, and optionally let the model correct itself with max_corrections:

step1 <- extractor$run_sync(
  "Extract the metadata of the R package in the working directory.",
  type = metadata_type,
  validate = function(x) {
    if (nrow(x$exported_functions) > 0) {
      TRUE
    } else {
      "No exported functions found. Look for @export tags in R/."
    }
  },
  max_corrections = 1
)

Structured output covers validation in detail.

Next steps