Nothing
#' Read Dimensions CSV export
#'
#' Parses a CSV file exported from Dimensions into a standardized
#' bibliometric data frame.
#'
#' @param file Path to a Dimensions CSV export file.
#' @param encoding Character. File encoding. Default `"UTF-8"`.
#'
#' @return A data frame in the standard bibnets format: `id`, `title`,
#' `year`, `journal`, `doi`, `cited_by_count`, `abstract`, `type`,
#' plus list-columns `authors`, `references`, and `keywords`.
#' Dimensions-specific extras: `affiliations` (list-column),
#' `countries` (list-column).
#'
#' @export
#' @examples
#' f <- system.file("extdata", "dimensions_sample.csv", package = "bibnets")
#' data <- read_dimensions(f)
#' head(data[, c("id", "title", "year", "journal")])
read_dimensions <- function(file, encoding = "UTF-8") {
check_file(file)
## Dimensions CSVs include a one-line metadata header ("About the data: ...")
## before the actual column-name row. Detect and skip it.
first_line <- readLines(file, n = 1L, encoding = encoding, warn = FALSE)
skip_rows <- if (grepl("^\"?About the data", first_line)) 1L else 0L
raw <- utils::read.csv(file, stringsAsFactors = FALSE, fileEncoding = encoding,
check.names = FALSE, skip = skip_rows)
col_map <- list(
id = c("Publication ID", "Dimensions URL"),
title = c("Title"),
year = c("PubYear", "Publication Year", "Year"),
journal = c("Source title", "Source Title",
"Source title/Anthology title"),
doi = c("DOI"),
cited_by = c("Times cited", "Times Cited", "Citation Count"),
abstract = c("Abstract"),
type = c("Publication Type", "Document Type"),
authors = c("Authors"),
refs = c("Cited references", "Cited References", "References"),
affiliations = c("Authors Affiliations - Name of Research organization",
"Authors Affiliations Name of Research organization",
"Research Organizations - standardized"),
countries = c("Authors Affiliations - Country of Research organization",
"Authors Affiliations Country of Research organization",
"Countries of Research organization")
)
lower_names <- tolower(names(raw))
find_col <- function(candidates) {
idx <- match(tolower(candidates), lower_names)
idx <- idx[!is.na(idx)]
if (length(idx) == 0L) return(NA_character_)
names(raw)[idx[1L]]
}
get_col <- function(candidates, default = NA_character_) {
col_name <- find_col(candidates)
if (is.na(col_name)) return(rep(default, nrow(raw)))
raw[[col_name]]
}
## Build ID
id_raw <- get_col(col_map$id, NA_character_)
id <- ifelse(is.na(id_raw) | nchar(id_raw) == 0,
paste0("DIM", seq_len(nrow(raw))),
id_raw)
result <- data.frame(
id = id,
title = get_col(col_map$title),
year = as.integer(get_col(col_map$year, NA_integer_)),
journal = get_col(col_map$journal),
doi = get_col(col_map$doi),
cited_by_count = as.integer(get_col(col_map$cited_by, 0L)),
abstract = get_col(col_map$abstract),
type = get_col(col_map$type),
stringsAsFactors = FALSE
)
## Authors: semicolon-delimited
result$authors <- lapply(
split_field(get_col(col_map$authors), sep = ";"),
standardize_authors
)
## References: Dimensions uses semicolons before brackets or just semicolons
refs_raw <- get_col(col_map$refs)
result$references <- lapply(refs_raw, function(r) {
if (is.na(r) || nchar(trimws(r)) == 0) return(character(0))
r <- gsub("\\[([^]]+)\\]", "\\1", r)
standardize_refs(strsplit(r, ";")[[1]])
})
## Keywords: Dimensions may have "Fields of Study" or "Research Categories"
kw_raw <- get_col(c("Fields of Study", "Keywords", "RCDC Categories",
"Research Categories"))
result$keywords <- split_field(kw_raw, sep = ";")
## Source-specific extras — both are list-columns so institution_network()
## and country_network() can use them directly
result$affiliations <- split_field(get_col(col_map$affiliations), sep = ";")
result$countries <- split_field(get_col(col_map$countries), sep = ";")
result
}
Any scripts or data that you put into this service are public.
Add the following code to your website.
For more information on customizing the embed code, read Embedding Snippets.