#load csv file
library(tidyverse)
UNSpeechcodes<-read.csv("Econcriticsnew.csv")
UNSpeechcodes$Text<-as.character(UNSpeechcodes$Text)
library(quanteda)
#make a corpus of the Texts
UNSpeechcodes_gen<-select(UNSpeechcodes,"Doc","cnt", "year", "Text")#takes selected columns

corpus_UN<-corpus(UNSpeechcodes,docid_field="Doc",text_field="Text")

UNSpeechcodes_gen<-select(UNSpeechcodes,"Doc","Text")#takes selected columns
UNSpeechcodes_gen$Doc<-gsub(".*_","",UNSpeechcodes_gen$Doc)#removes country code from Doc column
UNSpeechcodes_gen$Doc<-gsub("\\-.*","",UNSpeechcodes_gen$Doc)#removes everything after year in Doc column

UNSpeechcodes_gen<-UNSpeechcodes_gen%>%#collapses the data frame by year
  group_by(Doc)%>%
  summarise_all(funs(toString))

UNSpeechcodes_gen$year<-UNSpeechcodes_gen$Doc#adds a new column of year that will become docvars

corpus_gen<-corpus(UNSpeechcodes_gen,docid_field="Doc",text_field="Text")#creates a corpus

my_dict<-dictionary(list(Debt=c("debt","indebted","HIPC","Baker Plan","Brady Plan"), 
                         Conditionality=c("conditionality","conditionalities","structural adjustment","austerity"),
                         Order=c("order","structural change","structure"), 
                         Neoliberal=c("neoliberal","neo-liberal", "Washington Consensus", "washington consensus", "neoliberalism","neo-liberalism"),
                         Investment=c("investment","corporation","corporations","multinationals","enterprise","enterprises"),
                         Tariffs=c("tariff","tariffs","trade barrier"),
                         Rights=c("right","rights"), 
                         ClimateChange=c("climate change","global warming"),
                         Poverty=c("poverty","food","poor","hunger","destitution"),
                         Aid=c("aid","assistance"),
                         Sovereignty=c("sovereign","sovereignty","interference","intervention","non-interference"),
                        Inequality=c("inequality","equity","distribution","distributive","inequalities"),
                        Power=c("power","powerful","powerless","weak","hegemon","hegemony")))

#sets a dictionary of the key words

dfm_keywords<-dfm(corpus_gen,remove=stopwords("english"),remove_punct=TRUE,dictionary=my_dict)#make a data frame matrix out of the corpus that specifically looks at the key workds

keywords<-convert(dfm_keywords,to="data.frame")#produces a dataframe of the DFM for the keywords
keywords<-keywords[-1,]#removes Liberia row

keytidy <-gather(keywords, topic , frequency , -document)

png("BubbleplotISQ.png", width = 6, height = 4, units = 'in', res = 1600)

ggplot(keytidy, aes(x = as.numeric(document), y = topic, color=ifelse(frequency==0, NA, frequency), size=ifelse(frequency==0, NA, frequency))) +
  geom_point()  +  theme_classic() +
  labs(x="Year",y="Topic") +  guides(color= guide_legend(), size=guide_legend())+
  scale_x_continuous(breaks=seq(1970, 2017, 5)) + labs(color="Frequency") + labs(size="Frequency")


dev.off()



