tune(search): improve semantic reranking quality

This commit is contained in:
yun-zhi-ztl 2026-03-15 17:09:46 +08:00
parent 3a5e8d711e
commit 090cb156ac
3 changed files with 31 additions and 4 deletions

View file

@ -10,6 +10,7 @@ import org.springframework.stereotype.Service;
public class HashingSearchEmbeddingService implements SearchEmbeddingService {
private static final Pattern TOKEN_SPLITTER = Pattern.compile("[^\\p{L}\\p{N}_]+");
private static final int DIMENSIONS = 64;
private static final double NGRAM_WEIGHT = 0.35D;
@Override
public String embed(String text) {
@ -47,16 +48,30 @@ public class HashingSearchEmbeddingService implements SearchEmbeddingService {
.map(String::trim)
.filter(token -> !token.isBlank())
.forEach(token -> {
int hash = token.hashCode();
int index = Math.floorMod(hash, DIMENSIONS);
double weight = 1D + Math.min(token.length(), 12) / 12D;
vector[index] += weight;
addTokenWeight(vector, token, 1D + Math.min(token.length(), 12) / 12D);
addCharacterNgrams(vector, token);
});
normalize(vector);
return vector;
}
private void addTokenWeight(double[] vector, String token, double weight) {
int hash = token.hashCode();
int index = Math.floorMod(hash, DIMENSIONS);
vector[index] += weight;
}
private void addCharacterNgrams(double[] vector, String token) {
if (token.length() < 3) {
return;
}
for (int i = 0; i <= token.length() - 3; i++) {
String trigram = token.substring(i, i + 3);
addTokenWeight(vector, trigram, NGRAM_WEIGHT);
}
}
private double[] parseVector(String serializedVector) {
String[] parts = serializedVector.split(",");
double[] vector = new double[parts.length];

View file

@ -75,6 +75,7 @@ public class PostgresFullTextIndexService implements SearchIndexService {
private String buildSemanticVector(SkillSearchDocument document) {
return searchEmbeddingService.embed(String.join("\n",
safe(document.title()),
safe(document.title()),
safe(document.summary()),
safe(document.keywords()),

View file

@ -26,4 +26,15 @@ class HashingSearchEmbeddingServiceTest {
assertThat(relevant).isGreaterThan(noisy);
}
@Test
void similarityShouldHandleSingularAndPluralForms() {
String pluralVector = service.embed("build strong habits with daily practice");
String unrelatedVector = service.embed("research company profiles on the web");
double pluralMatch = service.similarity("habit", pluralVector);
double unrelatedMatch = service.similarity("habit", unrelatedVector);
assertThat(pluralMatch).isGreaterThan(unrelatedMatch);
}
}