added dependency updates, remade stuff to updates. fixed removing sentences. added null checks
This commit is contained in:
@@ -42,8 +42,10 @@ public class Datahandler {
|
||||
private var pipelineSentimentAnnotationCache = HashMap<String, Annotation>()
|
||||
private var coreDocumentAnnotationCache: HashMap<String, CoreDocument>
|
||||
private var jmweAnnotationCache = HashMap<String, Annotation>()
|
||||
private val stringCache = ArrayList<String>()
|
||||
private val nerModel = "edu/stanford/nlp/models/ner/english.all.3class.caseless.distsim.crf.ser.gz"
|
||||
private var stringCache = ArrayList<String>()
|
||||
|
||||
//private val nerModel = "edu/stanford/nlp/models/ner/english.all.3class.caseless.distsim.crf.ser.gz"
|
||||
private val nerModel = "edu/stanford/nlp/models/ner/english.all.3class.distsim.crf.ser.gz"
|
||||
private var tagger: MaxentTagger = MaxentTagger()
|
||||
private var gsf: GrammaticalStructureFactory
|
||||
private var classifier: AbstractSequenceClassifier<CoreLabel>
|
||||
@@ -99,8 +101,9 @@ public class Datahandler {
|
||||
|
||||
fun initiateGrammaticalStructureFactory(): GrammaticalStructureFactory {
|
||||
val options = arrayOf("-maxLength", "100")
|
||||
val lexParserEnglishRNN = "edu/stanford/nlp/models/lexparser/englishRNN.ser.gz"
|
||||
val lp = LexicalizedParser.loadModel(lexParserEnglishRNN, *options)
|
||||
//val lexParserEnglishRNN = "edu/stanford/nlp/models/lexparser/englishRNN.ser.gz"
|
||||
val lexParserEnglishPCFG = "edu/stanford/nlp/models/lexparser/englishPCFG.ser.gz"
|
||||
val lp = LexicalizedParser.loadModel(lexParserEnglishPCFG, *options)
|
||||
val tlp = lp.getOp().langpack()
|
||||
return tlp.grammaticalStructureFactory()
|
||||
}
|
||||
@@ -108,8 +111,10 @@ public class Datahandler {
|
||||
public fun pipeLineSetUp(): StanfordCoreNLP {
|
||||
val props = Properties()
|
||||
val shiftReduceParserPath = "edu/stanford/nlp/models/srparser/englishSR.ser.gz"
|
||||
val nerModel2 = "edu/stanford/nlp/models/ner/english.conll.4class.caseless.distsim.crf.ser.gz"
|
||||
val nerModel3 = "edu/stanford/nlp/models/ner/english.muc.7class.caseless.distsim.crf.ser.gz"
|
||||
//val nerModel2 = "edu/stanford/nlp/models/ner/english.conll.4class.caseless.distsim.crf.ser.gz"
|
||||
val nerModel2 = "edu/stanford/nlp/models/ner/english.conll.4class.distsim.crf.ser.gz"
|
||||
//val nerModel3 = "edu/stanford/nlp/models/ner/english.muc.7class.caseless.distsim.crf.ser.gz"
|
||||
val nerModel3 = "edu/stanford/nlp/models/ner/english.muc.7class.distsim.crf.ser.gz"
|
||||
props.setProperty("annotators", "tokenize,ssplit,pos,lemma,ner,parse")
|
||||
props.setProperty("parse.model", shiftReduceParserPath)
|
||||
props.setProperty("parse.maxlen", "90")
|
||||
@@ -129,12 +134,14 @@ public class Datahandler {
|
||||
|
||||
fun shiftReduceParserInitiate(): StanfordCoreNLP {
|
||||
val propsSentiment = Properties()
|
||||
val lexParserEnglishRNN = "edu/stanford/nlp/models/lexparser/englishRNN.ser.gz"
|
||||
//val lexParserEnglishRNN = "edu/stanford/nlp/models/lexparser/englishRNN.ser.gz"
|
||||
val lexParserEnglishPCFG = "edu/stanford/nlp/models/lexparser/englishPCFG.ser.gz"
|
||||
val sentimentModel = "edu/stanford/nlp/models/sentiment/sentiment.ser.gz"
|
||||
val taggerPath = "edu/stanford/nlp/models/pos-tagger/english-left3words/english-left3words-distsim.tagger"
|
||||
//val taggerPath = "edu/stanford/nlp/models/pos-tagger/english-left3words/english-left3words-distsim.tagger"
|
||||
val taggerPath = "edu/stanford/nlp/models/pos-tagger/english-left3words-distsim.tagger"
|
||||
val customStopWordList = "start,starts,period,periods,a,an,and,are,as,at,be,but,by,for,if,in,into,is,it,no,not,of," +
|
||||
"on,or,such,that,the,their,then,there,these,they,this,to,was,will,with"
|
||||
propsSentiment.setProperty("parse.model", lexParserEnglishRNN)
|
||||
propsSentiment.setProperty("parse.model", lexParserEnglishPCFG)
|
||||
propsSentiment.setProperty("sentiment.model", sentimentModel)
|
||||
propsSentiment.setProperty("parse.maxlen", "90")
|
||||
propsSentiment.setProperty("threads", "5")
|
||||
@@ -163,6 +170,8 @@ public class Datahandler {
|
||||
val arrayList = java.util.ArrayList<String>(stringCache)
|
||||
DataMapper.InsertMYSQLStrings(arrayList)
|
||||
DataMapper.checkStringsToDelete();
|
||||
stringCache = ArrayList<String>();
|
||||
initiateMYSQL();
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -81,9 +81,7 @@ public class DoStuff {
|
||||
}
|
||||
}
|
||||
}
|
||||
new Thread(() -> {
|
||||
datahandler.updateStringCache();
|
||||
}).start();
|
||||
datahandler.updateStringCache();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+25
-14
@@ -1772,6 +1772,9 @@ public class SentimentAnalyzerTest {
|
||||
}
|
||||
|
||||
private ArrayList<String> getnerEntities(CoreDocument coreDocument) {
|
||||
if (coreDocument == null || coreDocument.entityMentions() == null) {
|
||||
return new ArrayList<String>();
|
||||
}
|
||||
ArrayList<String> arrs = new ArrayList<>();
|
||||
for (CoreEntityMention em : coreDocument.entityMentions()) {
|
||||
if (!arrs.contains(em.text())) {
|
||||
@@ -1782,6 +1785,9 @@ public class SentimentAnalyzerTest {
|
||||
}
|
||||
|
||||
private ArrayList<String> getnerEntitiesType(CoreDocument coreDocument) {
|
||||
if (coreDocument == null || coreDocument.entityMentions() == null) {
|
||||
return new ArrayList<String>();
|
||||
}
|
||||
ArrayList<String> arrs = new ArrayList<>();
|
||||
for (CoreEntityMention em : coreDocument.entityMentions()) {
|
||||
if (!arrs.contains(em.entityType())) {
|
||||
@@ -1841,22 +1847,27 @@ public class SentimentAnalyzerTest {
|
||||
}
|
||||
|
||||
private ArrayList<String> getentityTokenTags(CoreDocument coreDocument) {
|
||||
if (coreDocument == null || coreDocument.entityMentions() == null) {
|
||||
return new ArrayList<String>();
|
||||
}
|
||||
ArrayList<String> arrs = new ArrayList<>();
|
||||
for (CoreEntityMention em : coreDocument.entityMentions()) {
|
||||
List<CoreLabel> tokens = em.tokens();
|
||||
String entityType = em.entityType();
|
||||
Double EntityConfidences = 0.0;
|
||||
Set<Map.Entry<String, Double>> entrySet = em.entityTypeConfidences().entrySet();
|
||||
for (Map.Entry<String, Double> entries : entrySet) {
|
||||
if (EntityConfidences < entries.getValue()) {
|
||||
EntityConfidences = entries.getValue();
|
||||
if (coreDocument != null) {
|
||||
for (CoreEntityMention em : coreDocument.entityMentions()) {
|
||||
List<CoreLabel> tokens = em.tokens();
|
||||
String entityType = em.entityType();
|
||||
Double EntityConfidences = 0.0;
|
||||
Set<Map.Entry<String, Double>> entrySet = em.entityTypeConfidences().entrySet();
|
||||
for (Map.Entry<String, Double> entries : entrySet) {
|
||||
if (EntityConfidences < entries.getValue()) {
|
||||
EntityConfidences = entries.getValue();
|
||||
}
|
||||
}
|
||||
}
|
||||
for (CoreLabel token : tokens) {
|
||||
if (token != null) {
|
||||
if (!arrs.contains(token.tag())) {
|
||||
if (entityType.equals("PERSON") && EntityConfidences > 0.80) {
|
||||
arrs.add(token.tag());
|
||||
for (CoreLabel token : tokens) {
|
||||
if (token != null) {
|
||||
if (!arrs.contains(token.tag())) {
|
||||
if (entityType.equals("PERSON") && EntityConfidences > 0.80) {
|
||||
arrs.add(token.tag());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user