R/read-ris.R

Defines functions read_ris

Documented in read_ris

#' Read an RIS file
#'
#' Parses a `.ris` file into a standardized bibliometric data frame.
#' Like BibTeX, standard RIS does not include cited references.
#'
#' @param file Path to a `.ris` file.
#' @param encoding Character. File encoding. Default `"UTF-8"`.
#'
#' @return A data frame in the standard bibnets format: `id`, `title`,
#'   `year`, `journal`, `doi`, `cited_by_count`, `abstract`, `type`,
#'   plus list-columns `authors`, `references` (typically empty for
#'   RIS), and `keywords`.
#'
#' @export
#' @examples
#' # Write a minimal RIS record to a temp file, then read it
#' ris <- "TY  - JOUR
#' AU  - Smith, J.
#' AU  - Jones, K.
#' TI  - Bibliometric networks
#' JO  - Test Journal
#' PY  - 2020
#' DO  - 10.1000/test
#' ER  - "
#' f <- tempfile(fileext = ".ris")
#' writeLines(ris, f)
#' data <- read_ris(f)
#' data[, c("id", "title", "year", "journal", "doi")]
#' unlink(f)
read_ris <- function(file, encoding = "UTF-8") {
  check_file(file)

  lines <- readLines(file, encoding = encoding, warn = FALSE)

  records <- list()
  current <- list()

  for (line in lines) {
    line <- trimws(line)
    if (nchar(line) == 0) next

    ## End of record
    if (grepl("^ER\\s*-", line)) {
      if (length(current) > 0) {
        records <- c(records, list(current))
      }
      current <- list()
      next
    }

    ## Tag line: "TY  - JOUR"
    if (grepl("^[A-Z][A-Z0-9]\\s+-\\s+", line)) {
      tag <- sub("\\s+-.*$", "", line)
      tag <- trimws(tag)
      value <- sub("^[A-Z][A-Z0-9]\\s+-\\s+", "", line)
      value <- trimws(value)
      current[[tag]] <- c(current[[tag]], value)
    }
  }

  if (length(current) > 0) {
    records <- c(records, list(current))
  }

  n <- length(records)
  if (n == 0) return(empty_biblio_df())

  get_ris <- function(rec, tags, collapse = NULL) {
    for (tag in tags) {
      val <- rec[[tag]]
      if (!is.null(val)) {
        if (!is.null(collapse)) return(paste(val, collapse = collapse))
        return(val[1])
      }
    }
    NA_character_
  }

  result <- data.frame(
    id = vapply(records, function(r) {
      id_val <- get_ris(r, c("DO", "AN", "ID"))
      if (is.na(id_val)) paste0("RIS", which(
        vapply(records, identical, logical(1), r)
      ))
      else id_val
    }, character(1)),
    title = vapply(records, function(r) get_ris(r, c("TI", "T1", "CT")),
                   character(1)),
    year = vapply(records, function(r) {
      y <- get_ris(r, c("PY", "Y1", "DA"))
      if (is.na(y)) return(NA_integer_)
      ## Extract 4-digit year from date strings
      m <- regmatches(y, regexpr("\\d{4}", y))
      if (length(m) > 0) as.integer(m[1]) else NA_integer_
    }, integer(1)),
    journal = vapply(records, function(r) {
      get_ris(r, c("JO", "JF", "T2", "JA"))
    }, character(1)),
    doi = vapply(records, function(r) get_ris(r, "DO"), character(1)),
    cited_by_count = NA_integer_,
    abstract = vapply(records, function(r) get_ris(r, c("AB", "N2")),
                      character(1)),
    type = vapply(records, function(r) get_ris(r, "TY"), character(1)),
    stringsAsFactors = FALSE
  )

  ## Authors (AU tag, repeatable)
  result$authors <- lapply(records, function(r) {
    au <- r[["AU"]]
    if (is.null(au)) au <- r[["A1"]]
    if (is.null(au)) return(character(0))
    standardize_authors(au)
  })

  ## No references in standard RIS
  result$references <- replicate(n, character(0), simplify = FALSE)

  ## Keywords (KW tag, repeatable)
  result$keywords <- lapply(records, function(r) {
    kw <- r[["KW"]]
    if (is.null(kw)) return(character(0))
    trimws(kw)
  })

  result
}

Try the bibnets package in your browser

Any scripts or data that you put into this service are public.

bibnets documentation built on June 19, 2026, 1:06 a.m.