Nothing
#' Run permutations
#'
#' This function defines the number of permutations (default = 50) and subsequently runs the permutations on an input dataframe. A minimum number of 10 permutations is recommended in order to avoid having very large or very small stability values due to the stochastic nature of the process. On a large dataset, increasing the number of permutations can considerably slow down the analysis.
#'
#' @importFrom magrittr %>%
#' @importFrom dplyr if_else
#' @import ggplot2
#' @importFrom utils read.table write.table
#' @importFrom grDevices pdf dev.off
#' @param input Input dataframe (a dataframe object).
#' @param replicates Number of permutation replicates to perform (an integer; default replicates=50).
#' @return A dataframe with the output values of the permutation analysis.
#'
#' @examples
#' data("coral_symbionts")
#' perm <- RunPerm(input = coral_symbionts,replicates = 50)
#' perm
#' @export
RunPerm <- function(input,replicates=50){
###############
# Upload data #
###############
if (!(inherits(input, "data.frame") || (is.character(input) && file.exists(input)))) {
stop("Error: Input must be a valid data frame or an existing file path.")
}
if (inherits(input, "data.frame")) {
message("input is a data frame")
df_name = deparse(substitute(input))
input[is.na(input)] <- 0
data <- input
df.name <- deparse(substitute(data))
colnames(data) = c("Host", colnames(data)[2:dim(data)[2]])
# Replace data larger than 1 to 1
if (max(unlist(data[,-c(1)])) > 1){
warning("DATA NOT ENCODED AS 0 AND 1. I AM GOING TO CONVERT VALUES LARGER THAN 1 to 1.")
replace_larger_1 <- function(x){
if_else(x > 1,1,x)
}
data <- data %>% dplyr::mutate_if(is.numeric, replace_larger_1)
}
} else{
message("input is a not data frame")
data = read.table(input, header = T, sep=c("\t",","), fill=TRUE) # Read input data file
df_name = input
colnames(data) = c("Host", colnames(data)[2:dim(data)[2]])
# Replace missing data by 0
data[is.na(data)] <- 0
# Replace data larger than 1 to 1
if (max(unlist(data[,-c(1)])) > 1){
warning("DATA NOT ENCODED AS 0 AND 1. I AM GOING TO CONVERT VALUES LARGER THAN 1 to 1.")
replace_larger_1 <- function(x){
if_else(x > 1,1,x)
}
data <- data %>% dplyr::mutate_if(is.numeric, replace_larger_1)
}
}
#######################################
# Set the number of replicates to use #
#######################################
repli = replicates # Set the number of replicates to perform
########################
# Print settings used #
########################
message("\n")
message("Running with the following settings:\n")
message(paste("Input name: ",df_name,sep=""),"\n")
message(paste("Number of replicates: ",repli,sep=""),"\n")
message("\n")
############################################################
# Set a few variables used later by the randomization loop #
############################################################
list <- list() # Create an empty list
a = 1 # Initialize a counter what will be used to append every new element to the list
Host = unique(data$Host) # Define unique species
Host = Host[!is.na(Host)] # Remove the lines in the input without a defined host species
species_counter = 0 # Initialize a counter what will count the number of species
########################
# Randomization loops #
########################
message("Starting permutations")
for (sp in Host){ # For loops over each species sp
Host_sp = data[data$Host==sp,] # Subset the element from the raw data specific to the species sp
colonies = dim(Host_sp)[1] # Count the total number of colonies for species sp
species_counter = species_counter + 1
for (taxa in 2:(dim(Host_sp)[2])){
taxa_name = names(Host_sp[,taxa, drop = FALSE])
taxa_size = length(Host_sp[Host_sp[taxa]==1,taxa]) # Count the total number of colonies with crabs for species sp
message(paste("Currently running species ",species_counter,": ", sp," ","taxa: ", taxa_name, sep=""))
if(taxa_size==0){
warning(paste(sp," has no ", taxa_name," I can't run the randomization", sep=""))
next
}
if((dim(Host_sp)[1] > 5)){ # If there are at least 5 symbionts
for (replicates in 1:repli){ # For loops to replicate
substract_taxa = Host_sp # Make sure to reinitialize the raw input for each replicates
Samples = 0
for (i in 1:dim(Host_sp)[1]-1){ # Loop over each individuals in the colonies of species sp
substract_taxa = dplyr::sample_n(substract_taxa,colonies-i,replace=FALSE) # Randomly remove 1 individual without replacement.
Num = (length(substract_taxa[substract_taxa[taxa]==1,taxa])/(colonies-i))*100 # Looks at the number of colonies with crabs / the total number of remaining colonies (in %) for each species sp.
list[[a]] = c(Num,taxa_name,i,replicates,sp,colonies-Samples) # Append the number of crabs counted, number of individuals removed, the replicate and the species to the list.
a = a + 1 # Increment by one the list so at the next iteration of the loop the information is stored in the element a + 1 of the list.
Samples = Samples + 1
}
}
}
}
message(paste("=====> ", sp," finished", sep=""))
}
###########################################################################
# Combine permutations results and make sure they are in the right format #
###########################################################################
permuted_output_df = do.call(rbind, list) # Combine the results
permuted_output_df = as.data.frame(permuted_output_df) # Results into a data frame format
colnames(permuted_output_df) = c("Prevalence","Taxa","Substract","replicates","Host_sp","colonies") # define column names
permuted_output_df$Prevalence = as.numeric(permuted_output_df$Prevalence) # Make sure the number of crabs is a numeric value
permuted_output_df$Substract = as.numeric(permuted_output_df$Substract) # Make sure the number of individuals removed is a numeric value
return(permuted_output_df)
}
Any scripts or data that you put into this service are public.
Add the following code to your website.
For more information on customizing the embed code, read Embedding Snippets.