R/tokenize.R

Defines functions tokenize get_data get_node_type get_node_value detect_node remove_indent remove_trailing_colon remove_comments remove_empty_lines parse_tag_line

NODE_REGEX <- paste0(
  "^(?!\\s+)(",
  paste0(
    c(
      "Feature:",
      "Scenario:", "Example:",
      "Scenarios:", "Examples:",
      "Scenario Outline:", "Scenario Template:",
      "Background:",
      "Given", "When", "Then", "Step"
    ),
    collapse = "|"
  ),
  ")(\\s+)?([:print:]*)?"
)

TAG_LINE_REGEX <- "^\\s*@"

parse_tag_line <- function(line) {
  # Tags can contain alphanumeric, underscore, hyphen, and dot
  tags <- regmatches(line, gregexpr("@[[:alnum:]_.-]+", line))[[1]]
  sub("^@", "", tags)
}

#' @importFrom stringr str_detect
remove_empty_lines <- function(x) {
  x[!str_detect(x, "^$")]
}

#' @importFrom stringr str_detect
remove_comments <- function(x) {
  x[!str_detect(x, "^\\s*#")]
}

#' @importFrom stringr str_remove_all
remove_trailing_colon <- function(x) {
  str_remove_all(x, ":$")
}

#' @importFrom stringr str_remove_all
remove_indent <- function(x) {
  indent <- getOption("cucumber.indent", default = "^\\s{2}")
  blocks <- docstring_blocks(x)
  x[blocks == 0] <- str_remove_all(x[blocks == 0], indent)
  # A docstring is dedented as a unit, by whatever its opening delimiter loses,
  # so that indentation relative to the delimiter is preserved
  for (block in seq_len(max(blocks, 0L))) {
    lines <- which(blocks == block)
    opening <- x[lines[1]]
    width <- nchar(opening) - nchar(str_remove_all(opening, indent))
    x[lines] <- str_remove_all(x[lines], paste0("^\\s{0,", width, "}"))
  }
  x
}

#' @importFrom stringr str_detect
detect_node <- function(x) {
  str_detect(x, NODE_REGEX)
}

#' @importFrom stringr str_match
get_node_value <- function(x) {
  str_match(x, NODE_REGEX)[4]
}

#' @importFrom stringr str_match
get_node_type <- function(x) {
  str_match(x, NODE_REGEX)[2]
}

get_data <- function(x) {
  if (length(x) == 0) {
    return(NULL)
  }
  x
}

#' @importFrom purrr map
tokenize <- function(x) {
  x <- normalize_feature(x)
  x <- remove_empty_lines(x)
  x <- remove_comments(x)
  is_tag_line <- grepl(TAG_LINE_REGEX, x)
  indices <- detect_node(x)
  if (sum(indices) == 0) {
    abort("Error tokenizing Gherkin, no keywords found")
  }
  cumulative <- cumsum(indices)
  groups <- seq_len(max(cumulative))
  groups |>
    map(\(ind) {
      group_positions <- which(cumulative == ind)
      text <- x[group_positions]

      # Collect tag lines immediately preceding this group
      first_pos <- group_positions[1]
      tags <- character(0)
      j <- first_pos - 1
      while (j >= 1 && is_tag_line[j]) {
        tags <- c(parse_tag_line(x[j]), tags)
        j <- j - 1
      }

      local_indices <- detect_node(text)
      type <- get_node_type(text[1]) |>
        remove_trailing_colon()
      value <- get_node_value(text[1])
      children <- text[!local_indices]
      children <- remove_indent(children)

      if (type %in% c("Feature", "Scenario", "Background", "Scenario Outline")) {
        pre_node <- children[!cumsum(detect_node(children))]
        return(
          new_token(
            type = type,
            value = value,
            tags = tags,
            children = tokenize(children),
            # Store free-form text in data, excluding tag lines
            data = get_data(pre_node[!grepl(TAG_LINE_REGEX, pre_node)])
          )
        )
      } else if (type %in% c("Step", "Given", "When", "Then")) {
        return(
          new_token(
            type = type,
            value = value,
            data = get_data(children)
          )
        )
      } else if (type == "Scenarios") {
        return(
          new_token(
            type = type,
            value = value,
            tags = tags,
            data = get_data(children[!grepl(TAG_LINE_REGEX, children)])
          )
        )
      }
    })
}

Try the cucumber package in your browser

Any scripts or data that you put into this service are public.

cucumber documentation built on Oct. 5, 2026, 9:07 a.m.