with annotaton caching stores results in the DB becomes Obsolete, quite some changes but overall improvements
This commit is contained in:
+82
-6
@@ -17,6 +17,8 @@ import edu.stanford.nlp.ling.TaggedWord;
|
||||
import edu.stanford.nlp.neural.rnn.RNNCoreAnnotations;
|
||||
import edu.stanford.nlp.parser.shiftreduce.ShiftReduceParser;
|
||||
import edu.stanford.nlp.pipeline.Annotation;
|
||||
import edu.stanford.nlp.pipeline.CoreDocument;
|
||||
import edu.stanford.nlp.pipeline.CoreEntityMention;
|
||||
import edu.stanford.nlp.pipeline.StanfordCoreNLP;
|
||||
import edu.stanford.nlp.process.CoreLabelTokenFactory;
|
||||
import edu.stanford.nlp.process.DocumentPreprocessor;
|
||||
@@ -38,6 +40,7 @@ import java.io.StringReader;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.OptionalDouble;
|
||||
import java.util.Set;
|
||||
@@ -62,7 +65,6 @@ public class SentimentAnalyzerTest implements Callable<SimilarityMatrix> {
|
||||
private SimilarityMatrix smxParam;
|
||||
private String str;
|
||||
private String str1;
|
||||
private ShiftReduceParser model;
|
||||
private MaxentTagger tagger;
|
||||
private GrammaticalStructureFactory gsf;
|
||||
private StanfordCoreNLP pipeline;
|
||||
@@ -74,13 +76,15 @@ public class SentimentAnalyzerTest implements Callable<SimilarityMatrix> {
|
||||
private Annotation pipelineAnnotation2;
|
||||
private Annotation pipelineAnnotation1Sentiment;
|
||||
private Annotation pipelineAnnotation2Sentiment;
|
||||
private CoreDocument pipelineCoreDcoument1;
|
||||
private CoreDocument pipelineCoreDcoument2;
|
||||
|
||||
public SentimentAnalyzerTest(String str, String str1, SimilarityMatrix smxParam, Annotation str1Annotation, Annotation str2Annotation,
|
||||
Annotation strPipeline1, Annotation strPipeline2, Annotation strPipeSentiment1, Annotation strPipeSentiment2) {
|
||||
Annotation strPipeline1, Annotation strPipeline2, Annotation strPipeSentiment1, Annotation strPipeSentiment2,
|
||||
CoreDocument pipelineCoreDcoument1, CoreDocument pipelineCoreDcoument2) {
|
||||
this.str = str;
|
||||
this.str1 = str1;
|
||||
this.smxParam = smxParam;
|
||||
this.model = Datahandler.getModel();
|
||||
this.tagger = Datahandler.getTagger();
|
||||
this.pipeline = Datahandler.getPipeline();
|
||||
this.pipelineSentiment = Datahandler.getPipelineSentiment();
|
||||
@@ -90,8 +94,10 @@ public class SentimentAnalyzerTest implements Callable<SimilarityMatrix> {
|
||||
this.jmweStrAnnotation2 = str2Annotation;
|
||||
this.pipelineAnnotation1 = strPipeline1;
|
||||
this.pipelineAnnotation2 = strPipeline2;
|
||||
this.pipelineAnnotation1Sentiment = strPipeSentiment1; //maybe process?
|
||||
this.pipelineAnnotation1Sentiment = strPipeSentiment1;
|
||||
this.pipelineAnnotation2Sentiment = strPipeSentiment2;
|
||||
this.pipelineCoreDcoument1 = pipelineCoreDcoument1;
|
||||
this.pipelineCoreDcoument2 = pipelineCoreDcoument2;
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -106,12 +112,14 @@ public class SentimentAnalyzerTest implements Callable<SimilarityMatrix> {
|
||||
= PTBTokenizer.factory(new CoreLabelTokenFactory(), "untokenizable=firstDelete");
|
||||
tokenizer.setTokenizerFactory(ptbTokenizerFactory);
|
||||
for (List<HasWord> sentence : tokenizer) {
|
||||
taggedwordlist1.add(model.apply(tagger.tagSentence(sentence)).taggedYield());
|
||||
taggedwordlist1.add(tagger.tagSentence(sentence));
|
||||
//taggedwordlist1.add(model.apply(tagger.tagSentence(sentence)).taggedYield());
|
||||
}
|
||||
tokenizer = new DocumentPreprocessor(new StringReader(str));
|
||||
tokenizer.setTokenizerFactory(ptbTokenizerFactory);
|
||||
for (List<HasWord> sentence : tokenizer) {
|
||||
taggedwordlist2.add(model.apply(tagger.tagSentence(sentence)).taggedYield());
|
||||
taggedwordlist2.add(tagger.tagSentence(sentence));
|
||||
//taggedwordlist2.add(model.apply(tagger.tagSentence(sentence)).taggedYield());
|
||||
}
|
||||
int counter = 0;
|
||||
int counter1 = 0;
|
||||
@@ -817,6 +825,74 @@ public class SentimentAnalyzerTest implements Callable<SimilarityMatrix> {
|
||||
double SentenceScoreDiff = leven.computeLevenshteinDistance();
|
||||
SentenceScoreDiff *= 15;
|
||||
score -= SentenceScoreDiff;
|
||||
ConcurrentMap<Integer, String> nerEntities1 = new MapMaker().concurrencyLevel(2).makeMap();
|
||||
ConcurrentMap<Integer, String> nerEntities2 = new MapMaker().concurrencyLevel(2).makeMap();
|
||||
ConcurrentMap<Integer, String> nerEntities3 = new MapMaker().concurrencyLevel(2).makeMap();
|
||||
ConcurrentMap<Integer, String> nerEntities4 = new MapMaker().concurrencyLevel(2).makeMap();
|
||||
ConcurrentMap<Integer, String> nerEntityTokenTags1 = new MapMaker().concurrencyLevel(2).makeMap();
|
||||
ConcurrentMap<Integer, String> nerEntityTokenTags2 = new MapMaker().concurrencyLevel(2).makeMap();
|
||||
for (CoreEntityMention em : pipelineCoreDcoument1.entityMentions()) {
|
||||
Set<Map.Entry<String, Double>> entrySet = em.entityTypeConfidences().entrySet();
|
||||
String entityType = em.entityType();
|
||||
Double EntityConfidences = 0.0;
|
||||
for (Map.Entry<String, Double> entries : entrySet) {
|
||||
EntityConfidences = entries.getValue();
|
||||
}
|
||||
List<CoreLabel> tokens = em.tokens();
|
||||
for (CoreLabel token : tokens) {
|
||||
if (!nerEntityTokenTags1.values().contains(token.tag())) {
|
||||
if (entityType.equals("PERSON") && EntityConfidences < 0.80) {
|
||||
score -= 6000;
|
||||
} else {
|
||||
nerEntityTokenTags1.put(nerEntityTokenTags1.size() + 1, token.tag());
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!nerEntities1.values().contains(em.text())) {
|
||||
nerEntities1.put(nerEntities1.size() + 1, em.text());
|
||||
nerEntities3.put(nerEntities3.size() + 1, em.entityType());
|
||||
}
|
||||
}
|
||||
for (CoreEntityMention em : pipelineCoreDcoument2.entityMentions()) {
|
||||
Set<Map.Entry<String, Double>> entrySet = em.entityTypeConfidences().entrySet();
|
||||
String entityType = em.entityType();
|
||||
Double EntityConfidences = 0.0;
|
||||
for (Map.Entry<String, Double> entries : entrySet) {
|
||||
EntityConfidences = entries.getValue();
|
||||
}
|
||||
List<CoreLabel> tokens = em.tokens();
|
||||
for (CoreLabel token : tokens) {
|
||||
if (!nerEntityTokenTags2.values().contains(token.tag())) {
|
||||
if (entityType.equals("PERSON") && EntityConfidences < 0.80) {
|
||||
score -= 6000;
|
||||
} else {
|
||||
nerEntityTokenTags2.put(nerEntityTokenTags2.size() + 1, token.tag());
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!nerEntities2.values().contains(em.text())) {
|
||||
nerEntities2.put(nerEntities2.size() + 1, em.text());
|
||||
nerEntities4.put(nerEntities4.size() + 1, em.entityType());
|
||||
}
|
||||
}
|
||||
for (String strEnts1 : nerEntities1.values()) {
|
||||
Collection<String> values = nerEntities2.values();
|
||||
for (String strEnts2 : values) {
|
||||
if (strEnts1.equalsIgnoreCase(strEnts2)) {
|
||||
score += 7500;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (String strEnts1 : nerEntities3.values()) {
|
||||
if (nerEntities4.values().contains(strEnts1)) {
|
||||
score += 3500;
|
||||
}
|
||||
}
|
||||
for (String strToken : nerEntityTokenTags1.values()) {
|
||||
if (nerEntityTokenTags2.values().contains(strToken)) {
|
||||
score += 2500;
|
||||
}
|
||||
}
|
||||
} catch (Exception ex) {
|
||||
System.out.println("SENTIMENT stacktrace Overall catch: " + ex.getMessage() + "\n");
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user