mirror of
https://github.com/iflytek/skillhub.git
synced 2026-10-07 02:57:51 +00:00
feat(search): improve Chinese PostgreSQL full-text search
This commit is contained in:
parent
f86f04a2d3
commit
b10a9f76db
7 changed files with 206 additions and 15 deletions
|
|
@ -27,6 +27,11 @@
|
|||
<groupId>org.springframework</groupId>
|
||||
<artifactId>spring-context</artifactId>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>io.github.arbing</groupId>
|
||||
<artifactId>jieba-analysis</artifactId>
|
||||
<version>1.0.3.1</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.springframework.boot</groupId>
|
||||
<artifactId>spring-boot-starter-test</artifactId>
|
||||
|
|
|
|||
|
|
@ -0,0 +1,79 @@
|
|||
package com.iflytek.skillhub.search;
|
||||
|
||||
import com.huaban.analysis.jieba.JiebaSegmenter;
|
||||
import com.huaban.analysis.jieba.JiebaSegmenter.SegMode;
|
||||
import com.huaban.analysis.jieba.SegToken;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
/**
|
||||
* Normalizes mixed Chinese and ASCII search text into stable token sequences for
|
||||
* both indexing and querying while keeping the raw phrase available for
|
||||
* substring fallback matching.
|
||||
*/
|
||||
@Component
|
||||
public class SearchTextTokenizer {
|
||||
private static final Pattern WHITESPACE = Pattern.compile("\\s+");
|
||||
private static final Pattern ASCII_TOKEN = Pattern.compile("[\\p{ASCII}]+");
|
||||
|
||||
private final JiebaSegmenter jiebaSegmenter = new JiebaSegmenter();
|
||||
|
||||
public List<String> tokenizeForIndex(String text) {
|
||||
return tokenize(text, SegMode.INDEX);
|
||||
}
|
||||
|
||||
public List<String> tokenizeForQuery(String text) {
|
||||
return tokenize(text, SegMode.SEARCH);
|
||||
}
|
||||
|
||||
public String enrichForIndex(String rawText) {
|
||||
String normalized = normalize(rawText);
|
||||
if (normalized == null) {
|
||||
return "";
|
||||
}
|
||||
|
||||
LinkedHashSet<String> parts = new LinkedHashSet<>();
|
||||
parts.add(normalized);
|
||||
parts.addAll(tokenizeForIndex(normalized));
|
||||
return String.join(" ", parts);
|
||||
}
|
||||
|
||||
private List<String> tokenize(String text, SegMode mode) {
|
||||
String normalized = normalize(text);
|
||||
if (normalized == null) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
Set<String> tokens = new LinkedHashSet<>();
|
||||
for (SegToken token : jiebaSegmenter.process(normalized, mode)) {
|
||||
String normalizedToken = normalizeToken(token.word);
|
||||
if (normalizedToken != null) {
|
||||
tokens.add(normalizedToken);
|
||||
}
|
||||
}
|
||||
return List.copyOf(tokens);
|
||||
}
|
||||
|
||||
private String normalize(String value) {
|
||||
if (value == null) {
|
||||
return null;
|
||||
}
|
||||
String normalized = WHITESPACE.matcher(value.trim()).replaceAll(" ");
|
||||
return normalized.isBlank() ? null : normalized;
|
||||
}
|
||||
|
||||
private String normalizeToken(String token) {
|
||||
String normalized = normalize(token);
|
||||
if (normalized == null) {
|
||||
return null;
|
||||
}
|
||||
if (ASCII_TOKEN.matcher(normalized).matches()) {
|
||||
normalized = normalized.toLowerCase(Locale.ROOT);
|
||||
}
|
||||
return normalized;
|
||||
}
|
||||
}
|
||||
|
|
@ -6,6 +6,7 @@ import com.iflytek.skillhub.search.SearchEmbeddingService;
|
|||
import com.iflytek.skillhub.search.SearchQuery;
|
||||
import com.iflytek.skillhub.search.SearchQueryService;
|
||||
import com.iflytek.skillhub.search.SearchResult;
|
||||
import com.iflytek.skillhub.search.SearchTextTokenizer;
|
||||
import jakarta.persistence.EntityManager;
|
||||
import jakarta.persistence.Query;
|
||||
import java.util.Comparator;
|
||||
|
|
@ -17,7 +18,6 @@ import org.springframework.beans.factory.annotation.Value;
|
|||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* PostgreSQL-backed implementation of {@link SearchQueryService}.
|
||||
|
|
@ -28,7 +28,6 @@ import java.util.regex.Pattern;
|
|||
*/
|
||||
@Service
|
||||
public class PostgresFullTextQueryService implements SearchQueryService {
|
||||
private static final Pattern QUERY_TERM_SPLITTER = Pattern.compile("[^\\p{L}\\p{N}_]+");
|
||||
private static final int MAX_QUERY_TERMS = 8;
|
||||
private static final int SHORT_PREFIX_LENGTH = 2;
|
||||
private static final String TITLE_VECTOR_SQL = "to_tsvector('simple', coalesce(title, ''))";
|
||||
|
|
@ -37,19 +36,21 @@ public class PostgresFullTextQueryService implements SearchQueryService {
|
|||
private final EntityManager entityManager;
|
||||
private final SkillSearchDocumentJpaRepository searchDocumentRepository;
|
||||
private final SearchEmbeddingService searchEmbeddingService;
|
||||
private final SearchTextTokenizer searchTextTokenizer;
|
||||
private final boolean semanticEnabled;
|
||||
private final double semanticWeight;
|
||||
private final int candidateMultiplier;
|
||||
private final int maxCandidates;
|
||||
|
||||
public PostgresFullTextQueryService(EntityManager entityManager) {
|
||||
this(entityManager, null, null, false, 0.35D, 8, 120);
|
||||
this(entityManager, null, null, new SearchTextTokenizer(), false, 0.35D, 8, 120);
|
||||
}
|
||||
|
||||
@Autowired
|
||||
public PostgresFullTextQueryService(EntityManager entityManager,
|
||||
SkillSearchDocumentJpaRepository searchDocumentRepository,
|
||||
SearchEmbeddingService searchEmbeddingService,
|
||||
SearchTextTokenizer searchTextTokenizer,
|
||||
@Value("${skillhub.search.semantic.enabled:true}") boolean semanticEnabled,
|
||||
@Value("${skillhub.search.semantic.weight:0.35}") double semanticWeight,
|
||||
@Value("${skillhub.search.semantic.candidate-multiplier:8}") int candidateMultiplier,
|
||||
|
|
@ -57,6 +58,7 @@ public class PostgresFullTextQueryService implements SearchQueryService {
|
|||
this.entityManager = entityManager;
|
||||
this.searchDocumentRepository = searchDocumentRepository;
|
||||
this.searchEmbeddingService = searchEmbeddingService;
|
||||
this.searchTextTokenizer = searchTextTokenizer;
|
||||
this.semanticEnabled = semanticEnabled;
|
||||
this.semanticWeight = semanticWeight;
|
||||
this.candidateMultiplier = candidateMultiplier;
|
||||
|
|
@ -74,7 +76,7 @@ public class PostgresFullTextQueryService implements SearchQueryService {
|
|||
boolean hasKeyword = normalizedKeyword != null;
|
||||
boolean hasTsQuery = tsQuery != null;
|
||||
boolean useRelevanceOrdering = "relevance".equals(query.sortBy()) && hasKeyword;
|
||||
boolean useShortPrefixTitleSearch = hasTsQuery && normalizedKeyword.length() <= SHORT_PREFIX_LENGTH;
|
||||
boolean useShortPrefixTitleSearch = hasTsQuery && isShortAsciiPrefixSearch(normalizedKeyword);
|
||||
boolean useSemanticRerank = semanticEnabled
|
||||
&& hasKeyword
|
||||
&& "relevance".equals(query.sortBy())
|
||||
|
|
@ -299,10 +301,7 @@ public class PostgresFullTextQueryService implements SearchQueryService {
|
|||
return null;
|
||||
}
|
||||
|
||||
List<String> terms = QUERY_TERM_SPLITTER.splitAsStream(keyword.toLowerCase())
|
||||
.map(String::trim)
|
||||
.filter(term -> !term.isBlank())
|
||||
.distinct()
|
||||
List<String> terms = searchTextTokenizer.tokenizeForQuery(keyword).stream()
|
||||
.limit(MAX_QUERY_TERMS)
|
||||
.toList();
|
||||
|
||||
|
|
@ -319,13 +318,25 @@ public class PostgresFullTextQueryService implements SearchQueryService {
|
|||
}
|
||||
|
||||
return tsQueryTerms.stream()
|
||||
.map(term -> term + ":*")
|
||||
.map(term -> usePrefixMatch(term) ? term + ":*" : term)
|
||||
.reduce((left, right) -> left + " & " + right)
|
||||
.orElse(null);
|
||||
}
|
||||
|
||||
private boolean isTsQueryCompatibleTerm(String term) {
|
||||
return term.chars().anyMatch(ch -> Character.isLetter(ch) || ch == '_');
|
||||
return term.chars().anyMatch(ch -> Character.isLetter(ch) || Character.isIdeographic(ch) || ch == '_');
|
||||
}
|
||||
|
||||
private boolean usePrefixMatch(String term) {
|
||||
return term.chars().allMatch(ch -> ch < 128) && term.chars().anyMatch(Character::isLetter);
|
||||
}
|
||||
|
||||
private boolean isShortAsciiPrefixSearch(String keyword) {
|
||||
if (keyword == null || keyword.length() > SHORT_PREFIX_LENGTH) {
|
||||
return false;
|
||||
}
|
||||
List<String> terms = searchTextTokenizer.tokenizeForQuery(keyword);
|
||||
return !terms.isEmpty() && terms.stream().allMatch(this::usePrefixMatch);
|
||||
}
|
||||
|
||||
private record RankedSkill(Long skillId, double score) {
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ import com.iflytek.skillhub.domain.skill.SkillVersion;
|
|||
import com.iflytek.skillhub.domain.skill.SkillVersionRepository;
|
||||
import com.iflytek.skillhub.search.SearchIndexService;
|
||||
import com.iflytek.skillhub.search.SearchRebuildService;
|
||||
import com.iflytek.skillhub.search.SearchTextTokenizer;
|
||||
import com.iflytek.skillhub.search.SkillSearchDocument;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
|
|
@ -37,17 +38,20 @@ public class PostgresSearchRebuildService implements SearchRebuildService {
|
|||
private final NamespaceRepository namespaceRepository;
|
||||
private final SkillVersionRepository skillVersionRepository;
|
||||
private final SearchIndexService searchIndexService;
|
||||
private final SearchTextTokenizer searchTextTokenizer;
|
||||
private final ObjectMapper objectMapper;
|
||||
|
||||
public PostgresSearchRebuildService(
|
||||
SkillRepository skillRepository,
|
||||
NamespaceRepository namespaceRepository,
|
||||
SkillVersionRepository skillVersionRepository,
|
||||
SearchIndexService searchIndexService) {
|
||||
SearchIndexService searchIndexService,
|
||||
SearchTextTokenizer searchTextTokenizer) {
|
||||
this.skillRepository = skillRepository;
|
||||
this.namespaceRepository = namespaceRepository;
|
||||
this.skillVersionRepository = skillVersionRepository;
|
||||
this.searchIndexService = searchIndexService;
|
||||
this.searchTextTokenizer = searchTextTokenizer;
|
||||
this.objectMapper = new ObjectMapper();
|
||||
}
|
||||
|
||||
|
|
@ -94,8 +98,8 @@ public class PostgresSearchRebuildService implements SearchRebuildService {
|
|||
.ifPresent(frontmatter -> appendFrontmatter(frontmatter, keywords, searchParts));
|
||||
|
||||
return new SearchIndexPayload(
|
||||
String.join(", ", keywords),
|
||||
String.join(" ", searchParts).trim()
|
||||
searchTextTokenizer.enrichForIndex(String.join(" ", keywords)),
|
||||
searchTextTokenizer.enrichForIndex(String.join(" ", searchParts).trim())
|
||||
);
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,24 @@
|
|||
package com.iflytek.skillhub.search;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
|
||||
class SearchTextTokenizerTest {
|
||||
|
||||
private final SearchTextTokenizer tokenizer = new SearchTextTokenizer();
|
||||
|
||||
@Test
|
||||
void enrichForIndexShouldKeepRawPhraseAndAddSegmentedTokens() {
|
||||
String enriched = tokenizer.enrichForIndex("中文技能搜索");
|
||||
|
||||
assertThat(enriched).contains("中文技能搜索");
|
||||
assertThat(tokenizer.tokenizeForIndex("中文技能搜索")).contains("中文", "技能", "搜索");
|
||||
}
|
||||
|
||||
@Test
|
||||
void tokenizeForQueryShouldNormalizeAsciiTerms() {
|
||||
assertThat(tokenizer.tokenizeForQuery("OpenAI Agent"))
|
||||
.containsExactly("openai", "agent");
|
||||
}
|
||||
}
|
||||
|
|
@ -4,6 +4,7 @@ import com.iflytek.skillhub.infra.jpa.SkillSearchDocumentEntity;
|
|||
import com.iflytek.skillhub.infra.jpa.SkillSearchDocumentJpaRepository;
|
||||
import com.iflytek.skillhub.search.HashingSearchEmbeddingService;
|
||||
import com.iflytek.skillhub.search.SearchQuery;
|
||||
import com.iflytek.skillhub.search.SearchTextTokenizer;
|
||||
import com.iflytek.skillhub.search.SearchVisibilityScope;
|
||||
import jakarta.persistence.EntityManager;
|
||||
import jakarta.persistence.Query;
|
||||
|
|
@ -142,6 +143,36 @@ class PostgresFullTextQueryServiceTest {
|
|||
assertThat(sqlCaptor.getAllValues().getFirst()).doesNotContain("ORDER BY ORDER BY");
|
||||
}
|
||||
|
||||
@Test
|
||||
void shortChineseKeywordsShouldStillUseSearchVectorInsteadOfTitlePrefixOptimization() {
|
||||
EntityManager entityManager = mock(EntityManager.class);
|
||||
Query nativeQuery = mock(Query.class);
|
||||
Query countQuery = mock(Query.class);
|
||||
when(entityManager.createNativeQuery(anyString()))
|
||||
.thenReturn(nativeQuery)
|
||||
.thenReturn(countQuery);
|
||||
when(nativeQuery.setParameter(anyString(), org.mockito.ArgumentMatchers.any())).thenReturn(nativeQuery);
|
||||
when(countQuery.setParameter(anyString(), org.mockito.ArgumentMatchers.any())).thenReturn(countQuery);
|
||||
when(nativeQuery.getResultList()).thenReturn(List.of());
|
||||
when(countQuery.getSingleResult()).thenReturn(0L);
|
||||
|
||||
PostgresFullTextQueryService service = new PostgresFullTextQueryService(entityManager);
|
||||
|
||||
service.search(new SearchQuery(
|
||||
"中文",
|
||||
null,
|
||||
new SearchVisibilityScope(null, Set.of(), Set.of()),
|
||||
"relevance",
|
||||
0,
|
||||
20
|
||||
));
|
||||
|
||||
var sqlCaptor = org.mockito.ArgumentCaptor.forClass(String.class);
|
||||
verify(entityManager, org.mockito.Mockito.times(2)).createNativeQuery(sqlCaptor.capture());
|
||||
assertThat(sqlCaptor.getAllValues().getFirst()).contains("d.search_vector @@ to_tsquery('simple', :tsQuery)");
|
||||
assertThat(sqlCaptor.getAllValues().getFirst()).doesNotContain("ts_rank_cd(to_tsvector('simple', coalesce(title, '')), to_tsquery('simple', :tsQuery))");
|
||||
}
|
||||
|
||||
@Test
|
||||
void multipleTermsShouldBuildPrefixQueryForEachLexeme() {
|
||||
EntityManager entityManager = mock(EntityManager.class);
|
||||
|
|
@ -170,6 +201,34 @@ class PostgresFullTextQueryServiceTest {
|
|||
verify(countQuery).setParameter("tsQuery", "self:* & improving:*");
|
||||
}
|
||||
|
||||
@Test
|
||||
void chineseKeywordsShouldUseSegmentedTermsWithoutAsciiPrefixSuffix() {
|
||||
EntityManager entityManager = mock(EntityManager.class);
|
||||
Query nativeQuery = mock(Query.class);
|
||||
Query countQuery = mock(Query.class);
|
||||
when(entityManager.createNativeQuery(anyString()))
|
||||
.thenReturn(nativeQuery)
|
||||
.thenReturn(countQuery);
|
||||
when(nativeQuery.setParameter(anyString(), org.mockito.ArgumentMatchers.any())).thenReturn(nativeQuery);
|
||||
when(countQuery.setParameter(anyString(), org.mockito.ArgumentMatchers.any())).thenReturn(countQuery);
|
||||
when(nativeQuery.getResultList()).thenReturn(List.of());
|
||||
when(countQuery.getSingleResult()).thenReturn(0L);
|
||||
|
||||
PostgresFullTextQueryService service = new PostgresFullTextQueryService(entityManager);
|
||||
|
||||
service.search(new SearchQuery(
|
||||
"中文技能搜索",
|
||||
null,
|
||||
new SearchVisibilityScope(null, Set.of(), Set.of()),
|
||||
"relevance",
|
||||
0,
|
||||
20
|
||||
));
|
||||
|
||||
verify(nativeQuery).setParameter("tsQuery", "中文 & 技能 & 搜索");
|
||||
verify(countQuery).setParameter("tsQuery", "中文 & 技能 & 搜索");
|
||||
}
|
||||
|
||||
@Test
|
||||
void numericKeywordsShouldFallbackToTitleLikeSearchWithoutTsQuery() {
|
||||
EntityManager entityManager = mock(EntityManager.class);
|
||||
|
|
@ -391,6 +450,7 @@ class PostgresFullTextQueryServiceTest {
|
|||
entityManager,
|
||||
repository,
|
||||
embeddingService,
|
||||
new SearchTextTokenizer(),
|
||||
true,
|
||||
0.6D,
|
||||
8,
|
||||
|
|
@ -430,6 +490,7 @@ class PostgresFullTextQueryServiceTest {
|
|||
entityManager,
|
||||
repository,
|
||||
embeddingService,
|
||||
new SearchTextTokenizer(),
|
||||
true,
|
||||
0.6D,
|
||||
8,
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ import com.iflytek.skillhub.domain.skill.SkillVersion;
|
|||
import com.iflytek.skillhub.domain.skill.SkillVersionRepository;
|
||||
import com.iflytek.skillhub.domain.skill.SkillVisibility;
|
||||
import com.iflytek.skillhub.search.SearchIndexService;
|
||||
import com.iflytek.skillhub.search.SearchTextTokenizer;
|
||||
import com.iflytek.skillhub.search.SkillSearchDocument;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.mockito.ArgumentCaptor;
|
||||
|
|
@ -64,7 +65,8 @@ class PostgresSearchRebuildServiceTest {
|
|||
skillRepository,
|
||||
namespaceRepository,
|
||||
skillVersionRepository,
|
||||
searchIndexService
|
||||
searchIndexService,
|
||||
new SearchTextTokenizer()
|
||||
);
|
||||
|
||||
service.rebuildBySkill(1L);
|
||||
|
|
@ -73,7 +75,12 @@ class PostgresSearchRebuildServiceTest {
|
|||
verify(searchIndexService).index(captor.capture());
|
||||
|
||||
SkillSearchDocument document = captor.getValue();
|
||||
assertThat(document.keywords()).isEqualTo("agentic, assistant, automation, workflow");
|
||||
assertThat(document.title()).isEqualTo("Smart Agent");
|
||||
assertThat(document.summary()).isEqualTo("Builds workflows");
|
||||
assertThat(document.keywords()).contains("agentic");
|
||||
assertThat(document.keywords()).contains("assistant");
|
||||
assertThat(document.keywords()).contains("automation");
|
||||
assertThat(document.keywords()).contains("workflow");
|
||||
assertThat(document.searchText()).contains("Smart Agent");
|
||||
assertThat(document.searchText()).contains("smart-agent");
|
||||
assertThat(document.searchText()).contains("Builds workflows");
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue