This commit is contained in:
zaCade
2019-03-02 15:10:46 +01:00
parent fc01f7fb43
commit 57bbd5d55e
62 changed files with 2233 additions and 0 deletions
@@ -0,0 +1,73 @@
package FunctionLayer.misc;
import edu.stanford.nlp.ling.HasWord;
import edu.stanford.nlp.ling.IndexedWord;
import edu.stanford.nlp.ling.TaggedWord;
import edu.stanford.nlp.ling.Word;
import edu.stanford.nlp.parser.lexparser.LexicalizedParser;
import edu.stanford.nlp.process.DocumentPreprocessor;
import edu.stanford.nlp.process.Tokenizer;
import edu.stanford.nlp.trees.GrammaticalStructure;
import edu.stanford.nlp.trees.GrammaticalStructureFactory;
import edu.stanford.nlp.trees.Tree;
import edu.stanford.nlp.trees.TreebankLanguagePack;
import edu.stanford.nlp.trees.TypedDependency;
import java.io.StringReader;
import java.util.ArrayList;
import java.util.List;
/*
* To change this license header, choose License Headers in Project Properties.
* To change this template file, choose Tools | Templates
* and open the template in the editor.
*/
/**
*
* @author install1
*/
public class SentimentAnalyzerTest {
public static SentimentAnalyzerTest instance = new SentimentAnalyzerTest();
public static String grammar = "edu/stanford/nlp/models/lexparser/englishPCFG.ser.gz";
public static String[] options = {"-maxLength", "80", "-retainTmpSubcategories"};
public static LexicalizedParser initiateLexicalizedParser() {
LexicalizedParser lp = LexicalizedParser.loadModel(grammar, options);
return lp;
}
public static TreebankLanguagePack initiateTreebankLanguagePack(LexicalizedParser lp) {
TreebankLanguagePack tlp = lp.getOp().langpack();
return tlp;
}
public double sentimentanalyzing(String str, String str1, double sreturn, LexicalizedParser lp, TreebankLanguagePack tlp) {
Iterable<List<? extends HasWord>> sentences;
Tokenizer<? extends HasWord> toke
= tlp.getTokenizerFactory().getTokenizer(new StringReader(str));
List<? extends HasWord> sentence = toke.tokenize();
String[] sent3 = {str1};
String[] tag3 = {"PRP", "MD", "VB", "PRP", "."}; // Parser gets second "can" wrong without help
List<TaggedWord> sentence2 = new ArrayList<>();
for (int i = 0; i < sent3.length; i++) {
sentence2.add(new TaggedWord(sent3[i], tag3[i]));
}
//parse.pennPrint();
List<List<? extends HasWord>> tmp
= new ArrayList<>();
tmp.add(sentence2);
tmp.add(sentence);
sentences = tmp;
for (List<? extends HasWord> sentence1 : sentences) {
Tree parse1 = lp.parse(sentence1);
GrammaticalStructureFactory gsf = tlp.grammaticalStructureFactory();
GrammaticalStructure gs = gsf.newGrammaticalStructure(parse1);
double score = parse1.score();
if (score > sreturn) {
//System.out.println("\n score : " + score + "\n");
sreturn = score;
}
}
return sreturn;
}
}
@@ -0,0 +1,66 @@
/*
* To change this license header, choose License Headers in Project Properties.
* To change this template file, choose Tools | Templates
* and open the template in the editor.
*/
package FunctionLayer.misc;
import DataLayer.DataMapper;
import FunctionLayer.CustomError;
import FunctionLayer.SimilarityMatrix;
import static FunctionLayer.MYSQLDatahandler.EXPIRE_TIME_IN_SECONDS;
import com.google.common.base.Stopwatch;
import com.google.common.collect.ArrayListMultimap;
import com.google.common.collect.MapMaker;
import com.google.common.collect.Multimap;
import java.io.IOException;
import java.sql.SQLException;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.concurrent.ConcurrentMap;
import java.util.concurrent.TimeUnit;
/**
*
* @author install1
*/
public class SentimentSimilarityCacheObsolete {
public static final long EXPIRE_TIME_IN_SECONDS = TimeUnit.SECONDS.convert(5, TimeUnit.HOURS);
private static ConcurrentMap<String, List<SimilarityMatrix>> SimilarityMatrixCache;
private static Stopwatch stopwatch;
public SentimentSimilarityCacheObsolete(ConcurrentMap<Integer, String> StringCache, Stopwatch stopwatch) {
this.stopwatch = Stopwatch.createUnstarted();
this.SimilarityMatrixCache = new MapMaker().concurrencyLevel(2).makeMap();
}
public void clearConCurrentMaps() {
SimilarityMatrixCache.clear();
}
private Multimap<String, SimilarityMatrix> getCache() throws SQLException, IOException, CustomError {
List<SimilarityMatrix> matrixlist;
matrixlist = DataMapper.getAllSementicMatrixes();
Multimap<String, SimilarityMatrix> LHM = ArrayListMultimap.create();
for (int i = 0; i < matrixlist.size(); i++) {
LHM.put(matrixlist.get(i).getPrimaryString(), matrixlist.get(i));
}
return LHM;
}
public static String getResponseStr(String strreturn) {
String str = "";
if (stopwatch.elapsed(TimeUnit.SECONDS) >= EXPIRE_TIME_IN_SECONDS || !stopwatch.isRunning()) {
if (!stopwatch.isRunning()) {
stopwatch.start();
} else {
stopwatch.reset();
}
}
return str;
}
}
@@ -0,0 +1,152 @@
/*
* To change this license header, choose License Headers in Project Properties.
* To change this template file, choose Tools | Templates
* and open the template in the editor.
https://github.com/DonatoMeoli/WS4J
*/
package FunctionLayer.misc;
import edu.cmu.lti.jawjaw.pobj.POS;
import edu.cmu.lti.lexical_db.ILexicalDatabase;
import edu.cmu.lti.lexical_db.NictWordNet;
import edu.cmu.lti.lexical_db.data.Concept;
import edu.cmu.lti.ws4j.Relatedness;
import edu.cmu.lti.ws4j.RelatednessCalculator;
import edu.cmu.lti.ws4j.impl.HirstStOnge;
import edu.cmu.lti.ws4j.impl.JiangConrath;
import edu.cmu.lti.ws4j.impl.LeacockChodorow;
import edu.cmu.lti.ws4j.impl.Lesk;
import edu.cmu.lti.ws4j.impl.Lin;
import edu.cmu.lti.ws4j.impl.Path;
import edu.cmu.lti.ws4j.impl.Resnik;
import edu.cmu.lti.ws4j.impl.WuPalmer;
/**
*
* @author install1
* https://www.programcreek.com/java-api-examples/?api=edu.cmu.lti.ws4j.RelatednessCalculator
* https://stackoverflow.com/questions/36300485/how-to-resolve-the-difference-between-the-values-attained-in-the-web-api-and-the
*/
public class WordNetSimalarityObsolete {
private static ILexicalDatabase db = new NictWordNet();
private static RelatednessCalculator rc1 = new WuPalmer(db);
private static RelatednessCalculator rc2 = new Resnik(db);
private static RelatednessCalculator rc3 = new JiangConrath(db);
private static RelatednessCalculator rc4 = new Lin(db);
private static RelatednessCalculator rc5 = new LeacockChodorow(db);
private static RelatednessCalculator rc6 = new Path(db);
private static RelatednessCalculator rc7 = new Lesk(db);
private static RelatednessCalculator rc8 = new HirstStOnge(db);
public static double SentenceMatcherSimilarityMatrix(String[] words1, String[] words2, double maxScore) {
{
double[][] s1 = getSimilarityMatrix(words1, words2, rc1);
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
if (s1[i][j] < maxScore && s1[i][j] > 0.0) {
System.out.print(s1[i][j] + "\t");
System.out.println("WuPalmer");
maxScore = s1[i][j];
}
}
}
}
{
double[][] s2 = getSimilarityMatrix(words1, words2, rc2);
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
if (s2[i][j] < maxScore && s2[i][j] > 0.0) {
System.out.println("Resnik");
System.out.print(s2[i][j] + "\t");
maxScore = s2[i][j];
}
}
}
}
{
double[][] s2 = getSimilarityMatrix(words1, words2, rc3);
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
if (s2[i][j] < maxScore && s2[i][j] > 0.0) {
System.out.println("JiangConrath");
System.out.print(s2[i][j] + "\t");
maxScore = s2[i][j];
}
}
}
}
{
double[][] s2 = getSimilarityMatrix(words1, words2, rc4);
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
if (s2[i][j] < maxScore && s2[i][j] > 0.0) {
System.out.println("Lin");
System.out.print(s2[i][j] + "\t");
maxScore = s2[i][j];
}
}
}
}
{
double[][] s2 = getSimilarityMatrix(words1, words2, rc5);
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
if (s2[i][j] < maxScore && s2[i][j] > 0.0) {
System.out.print(s2[i][j] + "\t");
System.out.println("LeacockChodrow");
maxScore = s2[i][j];
}
}
}
}
{
double[][] s2 = getSimilarityMatrix(words1, words2, rc6);
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
if (s2[i][j] < maxScore && s2[i][j] > 0.0) {
System.out.println("Path");
System.out.print(s2[i][j] + "\t");
maxScore = s2[i][j];
}
}
}
}
{
double[][] s2 = getSimilarityMatrix(words1, words2, rc7);
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
if (s2[i][j] < maxScore && s2[i][j] > 0.0) {
System.out.println("Lesk");
System.out.print(s2[i][j] + "\t");
maxScore = s2[i][j];
}
}
}
}
{
double[][] s2 = getSimilarityMatrix(words1, words2, rc8);
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
if (s2[i][j] < maxScore && s2[i][j] > 0.0) {
System.out.println("HirstStOnge");
System.out.print(s2[i][j] + "\t");
maxScore = s2[i][j];
}
}
}
}
return maxScore;
}
public static double[][] getSimilarityMatrix(String[] words1, String[] words2, RelatednessCalculator rc) {
double[][] result = new double[words1.length][words2.length];
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
double score = rc.calcRelatednessOfWords(words1[i], words2[j]);
result[i][j] = score;
}
}
return result;
}
}
@@ -0,0 +1,74 @@
/*
* To change this license header, choose License Headers in Project Properties.
* To change this template file, choose Tools | Templates
* and open the template in the editor.
*/
package FunctionLayer.misc;
import edu.cmu.lti.lexical_db.ILexicalDatabase;
import edu.cmu.lti.lexical_db.NictWordNet;
import edu.cmu.lti.ws4j.RelatednessCalculator;
import edu.cmu.lti.ws4j.impl.HirstStOnge;
import edu.cmu.lti.ws4j.impl.JiangConrath;
import edu.cmu.lti.ws4j.impl.LeacockChodorow;
import edu.cmu.lti.ws4j.impl.Lesk;
import edu.cmu.lti.ws4j.impl.Lin;
import edu.cmu.lti.ws4j.impl.Path;
import edu.cmu.lti.ws4j.impl.Resnik;
import edu.cmu.lti.ws4j.impl.WuPalmer;
import java.util.ArrayList;
import java.util.List;
/**
*
* @author install1
* http://ws4jdemo.appspot.com/?mode=s&s1=something+like+a+sentence&s2=should+not+be+like+the+first+one
*/
public class WordNetSimalarityTestObsolete {
private static ILexicalDatabase db = new NictWordNet();
private static RelatednessCalculator rc1 = new WuPalmer(db);
private static RelatednessCalculator rc2 = new Resnik(db);
private static RelatednessCalculator rc3 = new JiangConrath(db);
private static RelatednessCalculator rc4 = new Lin(db);
private static RelatednessCalculator rc5 = new LeacockChodorow(db);
private static RelatednessCalculator rc6 = new Path(db);
private static RelatednessCalculator rc7 = new Lesk(db);
private static RelatednessCalculator rc8 = new HirstStOnge(db);
public static double SentenceMatcherSimilarityMatrix(String[] words1, String[] words2, double maxScore) {
boolean initial = maxScore == 0.0;
List<RelatednessCalculator> RCList = new ArrayList();
RCList.add(rc1);
RCList.add(rc2);
RCList.add(rc3);
RCList.add(rc4);
RCList.add(rc5);
RCList.add(rc6);
RCList.add(rc7);
RCList.add(rc8);
for (int h = 0; h < RCList.size(); h++) {
double s1 = getSimilarityMatrix(words1, words2, RCList.get(h), maxScore, initial);
System.out.println("s1: " + String.format("%.0f", s1) + " \nmaxScore: " + maxScore);
if (s1 > 0.01 && (s1 < maxScore || initial)) {
maxScore = s1;
}
}
return maxScore;
}
public static double getSimilarityMatrix(String[] words1, String[] words2, RelatednessCalculator rc, double maxScore, boolean initial) {
double rtndouble = 0.01;
for (int i = 0; i < words1.length; i++) {
for (int j = 0; j < words2.length; j++) {
if (maxScore > rtndouble / words2.length || initial) {
rtndouble += (rc.calcRelatednessOfWords(words1[i], words2[j]));
//System.out.println("RelatednessCalculator: " + rc.toString());
} else {
return rtndouble;
}
}
}
return rtndouble;
}
}
@@ -0,0 +1,58 @@
/*
* To change this license header, choose License Headers in Project Properties.
* To change this template file, choose Tools | Templates
* and open the template in the editor.
*/
package FunctionLayer.misc;
/**
*
* @author install1
*/
public class notes {
/*
/*
ILexicalDatabase db = new NictWordNet();
RelatednessCalculator lesk = new Lesk(db);
POS posWord1 = POS.n;
POS posWord2 = POS.n;
double maxScore = 0;
WS4JConfiguration.getInstance().setMFS(true);
List<Concept> synsets1 = (List<Concept>) db.getAllConcepts(strreturn, posWord1.name());
for (int i = 0; i < allStringValuesPresent.size(); i++) {
List<Concept> synsets2 = (List<Concept>) db.getAllConcepts(allStringValuesPresent.get(i), posWord2.name());
for (Concept synset1 : synsets1) {
for (Concept synset2 : synsets2) {
Relatedness relatedness = lesk.calcRelatednessOfSynset(synset1, synset2);
double score = relatedness.getScore();
if (score > maxScore) {
maxScore = score;
index = i;
}
}
}
}
private static RelatednessCalculator[] rcs;
static {
WS4JConfiguration.getInstance().setMemoryDB(false);
WS4JConfiguration.getInstance().setMFS(true);
ILexicalDatabase db = new MITWordNet();
rcs = new RelatednessCalculator[]{
new HirstStOnge(db), new LeacockChodorow(db), new Lesk(db), new WuPalmer(db),
new Resnik(db), new JiangConrath(db), new Lin(db), new Path(db)
};
}
https://github.com/DonatoMeoli/WS4J
https://www.programcreek.com/2014/01/calculate-words-similarity-using-wordnet-in-java/
*/
/*
//available options of metrics
private static RelatednessCalculator[] rcs = { new HirstStOnge(db),
new LeacockChodorow(db), new Lesk(db), new WuPalmer(db),
new Resnik(db), new JiangConrath(db), new Lin(db), new Path(db) };
*/
}