Repository navigation
Expand file tree
/
Copy pathStringComparison.java
More file actions
64 lines (54 loc) · 2.31 KB
/
Copy pathStringComparison.java
File metadata and controls
64 lines (54 loc) · 2.31 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
package by.andd3dfx.string;
import org.apache.commons.text.similarity.JaroWinklerSimilarity;
import org.apache.commons.text.similarity.LevenshteinDistance;
import java.util.HashSet;
import java.util.Set;
import java.util.stream.IntStream;
/**
* <pre>
* String comparison methods:
*
* Dice-Sørensen coefficient, also known as:
* - Sørensen–Dice index
* - Sørensen index
* - Dice's coefficient
* - Dice similarity coefficient (DSC)
*
* Levenstein similarity
*
* Jaro-Winkler similarity
* </pre>
*
* @see <a href="https://en.wikipedia.org/wiki/Dice-S%C3%B8rensen_coefficient">Wiki page</a>
* @see <a href="https://youtu.be/i7O4R4gfZqE">Video solution</a>
* @see <a href="https://habr.com/ru/articles/671136/">Habr article</a>
* @see <a href="https://www.baeldung.com/cs/string-similarity-edit-distance">Baeldung article</a>
*/
public class StringComparison {
private static final LevenshteinDistance LEVENSHTEIN_DISTANCE = LevenshteinDistance.getDefaultInstance();
private static final JaroWinklerSimilarity JARO_WINKLER_SIMILARITY = new JaroWinklerSimilarity();
public static double serencen(String first, String second) {
return serencen(toBigram(first), toBigram(second));
}
public static double levensteinSimilarity(String first, String second) {
double levensteinDistance = LEVENSHTEIN_DISTANCE.apply(first, second);
return 1.0 - levensteinDistance / Math.max(first.length(), second.length());
}
public static double jaroWinklerSimilarity(String first, String second) {
return JARO_WINKLER_SIMILARITY.apply(first, second);
}
private static Set<String> toBigram(String text) {
if (text == null || text.length() < 2) {
return new HashSet<>();
}
// Создаем поток индексов от 0 до length - 2 и собираем биграммы в Set
return IntStream.range(0, text.length() - 1)
.mapToObj(i -> text.substring(i, i + 2))
.collect(HashSet::new, Set::add, Set::addAll);
}
public static <T> double serencen(Set<T> set1, Set<T> set2) {
var intersection = new HashSet<>(set1);
intersection.retainAll(set2);
return 2.0 * intersection.size() / (set1.size() + set2.size());
}
}