mirror of
https://github.com/iflytek/skillhub.git
synced 2026-10-07 02:57:51 +00:00
tune(search): improve semantic reranking quality
This commit is contained in:
parent
3a5e8d711e
commit
090cb156ac
3 changed files with 31 additions and 4 deletions
|
|
@ -10,6 +10,7 @@ import org.springframework.stereotype.Service;
|
|||
public class HashingSearchEmbeddingService implements SearchEmbeddingService {
|
||||
private static final Pattern TOKEN_SPLITTER = Pattern.compile("[^\\p{L}\\p{N}_]+");
|
||||
private static final int DIMENSIONS = 64;
|
||||
private static final double NGRAM_WEIGHT = 0.35D;
|
||||
|
||||
@Override
|
||||
public String embed(String text) {
|
||||
|
|
@ -47,16 +48,30 @@ public class HashingSearchEmbeddingService implements SearchEmbeddingService {
|
|||
.map(String::trim)
|
||||
.filter(token -> !token.isBlank())
|
||||
.forEach(token -> {
|
||||
int hash = token.hashCode();
|
||||
int index = Math.floorMod(hash, DIMENSIONS);
|
||||
double weight = 1D + Math.min(token.length(), 12) / 12D;
|
||||
vector[index] += weight;
|
||||
addTokenWeight(vector, token, 1D + Math.min(token.length(), 12) / 12D);
|
||||
addCharacterNgrams(vector, token);
|
||||
});
|
||||
|
||||
normalize(vector);
|
||||
return vector;
|
||||
}
|
||||
|
||||
private void addTokenWeight(double[] vector, String token, double weight) {
|
||||
int hash = token.hashCode();
|
||||
int index = Math.floorMod(hash, DIMENSIONS);
|
||||
vector[index] += weight;
|
||||
}
|
||||
|
||||
private void addCharacterNgrams(double[] vector, String token) {
|
||||
if (token.length() < 3) {
|
||||
return;
|
||||
}
|
||||
for (int i = 0; i <= token.length() - 3; i++) {
|
||||
String trigram = token.substring(i, i + 3);
|
||||
addTokenWeight(vector, trigram, NGRAM_WEIGHT);
|
||||
}
|
||||
}
|
||||
|
||||
private double[] parseVector(String serializedVector) {
|
||||
String[] parts = serializedVector.split(",");
|
||||
double[] vector = new double[parts.length];
|
||||
|
|
|
|||
|
|
@ -75,6 +75,7 @@ public class PostgresFullTextIndexService implements SearchIndexService {
|
|||
|
||||
private String buildSemanticVector(SkillSearchDocument document) {
|
||||
return searchEmbeddingService.embed(String.join("\n",
|
||||
safe(document.title()),
|
||||
safe(document.title()),
|
||||
safe(document.summary()),
|
||||
safe(document.keywords()),
|
||||
|
|
|
|||
|
|
@ -26,4 +26,15 @@ class HashingSearchEmbeddingServiceTest {
|
|||
|
||||
assertThat(relevant).isGreaterThan(noisy);
|
||||
}
|
||||
|
||||
@Test
|
||||
void similarityShouldHandleSingularAndPluralForms() {
|
||||
String pluralVector = service.embed("build strong habits with daily practice");
|
||||
String unrelatedVector = service.embed("research company profiles on the web");
|
||||
|
||||
double pluralMatch = service.similarity("habit", pluralVector);
|
||||
double unrelatedMatch = service.similarity("habit", unrelatedVector);
|
||||
|
||||
assertThat(pluralMatch).isGreaterThan(unrelatedMatch);
|
||||
}
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue