library(brms)
library(tidyverse)
library(stringr)
library(bayesplot)
library(ggrepel)
library(dplyr)
library(ggthemes)
library(data.table)
bayesplot::color_scheme_set("darkgray")
# Specific color for each bar? Use a well known palette
library(RColorBrewer)

#Read the data into a dataframe.
rawCSV <- read.csv("xp2022datascripts/example_data.csv", sep=";")
#The analysis and the graphs are incomplete with only example daata, however, all the 
#scripts work and show exactly how we cleaned and divided the data. 
glimpse(rawCSV)

onlyDateStage <- data.frame(rawCSV$date,rawCSV$stage,rawCSV$teamId)

glimpse(onlyDateStage)
x<-onlyDateStage

#The reorganization was the 1st of October 2020.
lst <- split(x,x$rawCSV.date<as.Date("2020-10-01"))
y<-as.data.frame(lst[1]) 
z<-as.data.frame(lst[2]) 
y
y <- y[(y$FALSE.rawCSV.date < "2021-04"), ]
z <- z[(z$TRUE.rawCSV.date > "2020-05"), ]
z

#Remove any duplicates
sortedy <- y[order(y$FALSE.rawCSV.teamId, decreasing=TRUE, y$FALSE.rawCSV.date), ]
keepy <- c(TRUE, head(sortedy$FALSE.rawCSV.teamId, -1) != tail(sortedy$FALSE.rawCSV.teamId,-1))
keepy

toKeepy<- sortedy[keepy, ] 
glimpse(toKeepy)
toKeepy


y<-toKeepy
y



#Clean the data from brackets etc
y$FALSE.rawCSV.stage<-gsub("\\[", "", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("\\]", "", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("1, 2", "Split", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("1, 3", "Split", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("1, 4", "Split", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("2, 3", "Split", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("2, 4", "Split", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("1, 3, 4", "Split", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("3, 4", "Split", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("Split, 4", "Split", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("Split, 3", "Split", y$FALSE.rawCSV.stage)
y$FALSE.rawCSV.stage<-gsub("Split, Split", "Split", y$FALSE.rawCSV.stage)

y <- table(y$FALSE.rawCSV.stage)
y1 <- c(y[5], y[1:4])
y<-y1

# create data frame
df <- data.frame(
  id = 1:5
  , Coolness_Level = 1:5
  , Coolness_Color = NA
  , stringsAsFactors = FALSE
)

# view data
df
# id Coolness_Level Coolness_Color
# 1  1              1             NA
# 2  2              2             NA
# 3  3              3             NA
# 4  4              4             NA
# 5  5              5             NA


# I want colors to progress
# from gray to dark blue
color.function <- colorRampPalette( c( "#CCCCCC" , "#104E8B" ) )

# decide how many groups I want, in this case 5
# so the end product will have 5 bars
color.ramp <- color.function( n = nrow( x = df ) )

# view colors
color.ramp
# [1] "#CCCCCC" "#9DACBB" "#6E8DAB" "#3F6D9B" "#104E8B"



#y <- table(y$FALSE.rawCSV.stage)
plot(y)
barplot(y[order(y, decreasing = TRUE)])
barplot(height=y, col=color.ramp, ylim=c(0,150) )


#Remove any duplicates.
sortedz <- z[order(z$TRUE.rawCSV.teamId, decreasing=TRUE, z$TRUE.rawCSV.date), ]
keepz <- c(TRUE, head(sortedz$TRUE.rawCSV.teamId, -1) != tail(sortedz$TRUE.rawCSV.teamId,-1))
keepz

toKeepz<- sortedz[keepz, ] 
glimpse(toKeepz)
toKeepz

#Clean the second data subset.
z$TRUE.rawCSV.stage<-gsub("\\[", "", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("\\]", "", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("1, 2", "Split", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("1, 3", "Split", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("1, 4", "Split", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("2, 3", "Split", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("2, 4", "Split", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("1, 3, 4", "Split", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("3, 4", "Split", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("Split, 4", "Split", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("Split, 3", "Split", z$TRUE.rawCSV.stage)
z$TRUE.rawCSV.stage<-gsub("Split, Split", "Split", z$TRUE.rawCSV.stage)
z
z <- table(z$TRUE.rawCSV.stage)

y2 <- c(z[5], z[1:4])
z<-y2


plot(z)
barplot(z[order(z, decreasing = TRUE)])
barplot(height=z, col=color.ramp, ylim=c(0,150))
#show number of teams at each category six months before and after the reorg. 
#The percentages was calculated using these tables.
y
z
