Repository navigation
Expand file tree
/
Copy pathModel_V.R
More file actions
126 lines (99 loc) · 3.33 KB
/
Copy pathModel_V.R
File metadata and controls
126 lines (99 loc) · 3.33 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
library(quanteda)
library(boot)
library(gdata)
library(stringr)
library(dplyr)
library(plyr)
library(data.table)
library(RTextTools)
library(e1071)
list_doc<- list.files(path = "txt_replicated_parsed/")
text=c()
#Read the documents
for (element in list_doc) {
file<- paste("txt_replicated_parsed/", element, sep="")
text<-c(text,readChar(file, file.info(file)$size))
}
dfText<- as.data.frame(as.numeric(gsub(".txt","", list_doc)))
dfText$text<-gsub("[[:punct:]]", " ", text)
colnames(dfText)<-c("name", "text")
#Paste the target variable
papers<-read.csv("replication_dataset.csv")
colnames(papers)<-c("name", "title", "target")
dfText<- join(dfText, papers, by="name")
#Creating dfm & tfidf matrix
dfm_Text<-dfm(dfText$text,verbose = FALSE,removePunctuation = TRUE,stem = TRUE,
removeNumbers = TRUE, toLower = TRUE, ignoredFeatures=stopwords(kind = "english"))
dfm_aux<-as.data.frame(dfm_Text)
auxCol<-colnames(dfm_Text)
dfm_Text<-dfm_Text[,auxCol[nchar(auxCol)>3 & nchar(auxCol)<10]]
dfm_tfid<-tfidf(dfm_Text)
#############################
################ Models
######## SVM
dtText<-subset(dfText, select = c("text","target"))
dim<-nrow(dfText)
#Sample size
smp_size <- floor(0.90 * dim)
svmDfm=c()
svmTfi=c()
for (i in seq(1:100)){
train_ind <- sample(seq_len(dim), size = smp_size)
#####Word frequency
#Train the model
svm.model<- svm(dfm_Text[train_ind, ], y = factor(dfText$target[train_ind]),
scale = TRUE, type = NULL, kernel ="linear")
#Predict the model
predicted<-predict(svm.model, dfm_Text[-train_ind,])
Y_predicted<-as.numeric(predicted)-1
Y_test<-dfText$target[-train_ind]
accuracy<-(1-sum(sign(abs(Y_predicted-Y_test)/2))/length(Y_test))
svmDfm<-c(svmDfm,accuracy)
######TFIDF
svm.model<- svm(dfm_tfid[train_ind, ], y = factor(dfText$target[train_ind]),
scale = TRUE, type = NULL, kernel ="linear")
predicted<-predict(svm.model, dfm_tfid[-train_ind,])
Y_predicted<-as.numeric(predicted)-1
Y_test<-dfText$target[-train_ind]
accuracy<-(1-sum(sign(abs(Y_predicted-Y_test)/2))/length(Y_test))
svmTfi<-c(svmTfi,accuracy)
}
####### Naive Bayes
#Convert the dfm to dataframe
df_Text<-as.data.frame(dfm_Text)
df_tfid<-as.data.frame(dfm_tfid)
#Sample size
smp_size <- floor(0.90 * nrow(df_Text))
NB=c()
for (i in seq(1:100)){
train_ind <- sample(seq_len(dim), size = smp_size)
#Train NB
clf<-textmodel_NB(df_Text[train_ind, ],
factor(dfText$target[train_ind]),smooth = 1, prior="uniform" )
#Predict the target variable
predicted<-predict(clf, newdata = as.matrix(df_Text[-train_ind, ])
, rescaling = "none", level = 0.95, verbose = TRUE)
#Predict
Y_predicted<-as.numeric(predicted$docs$predicted)
Y_predicted[Y_predicted==1]<-0
Y_predicted[Y_predicted==2]<-1
Y_test<-dfText$target[-train_ind]
accuracy<-(1-sum(sign(abs(Y_predicted-Y_test)/2))/length(Y_test))
NB<-c(NB,accuracy)
}
#####Data frames of the models.
dfDfm<-as.data.frame(svmDfm)
dfDfm$number<-"DFM SVM "
colnames(dfDfm)<-c("Mean","Model")
dfTfi<-as.data.frame(svmTfi)
dfTfi$number<-"TFIDF SVM "
colnames(dfTfi)<-c("Mean","Model")
NBDfm<-as.data.frame(NB)
NBDfm$number<-"NB"
colnames(NBDfm)<-c("Mean","Model")
#Boxplot
boxPlot<-rbind(dfDfm,dfTfi, NBDfm)
boxPlot$Model<-factor(boxPlot$Model)
boxplot(Mean~Model, data= boxPlot, xlab="Comparison betwen Models", ylab="Mean Accuracy" )
mean(NB)
sqrt(var(NB))