feat(search): improve Chinese PostgreSQL full-text search

This commit is contained in:
vsxd 2026-03-19 17:33:39 +08:00 • committed by Xudong Sun
parent f86f04a2d3
commit b10a9f76db
7 changed files with 206 additions and 15 deletions

View file

@ -27,6 +27,11 @@
<groupId>org.springframework</groupId>
<artifactId>spring-context</artifactId>
</dependency>
<dependency>
<groupId>io.github.arbing</groupId>
<artifactId>jieba-analysis</artifactId>
<version>1.0.3.1</version>
</dependency>
<dependency>
<groupId>org.springframework.boot</groupId>
<artifactId>spring-boot-starter-test</artifactId>

View file

@ -0,0 +1,79 @@
package com.iflytek.skillhub.search;
import com.huaban.analysis.jieba.JiebaSegmenter;
import com.huaban.analysis.jieba.JiebaSegmenter.SegMode;
import com.huaban.analysis.jieba.SegToken;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Locale;
import java.util.Set;
import java.util.regex.Pattern;
import org.springframework.stereotype.Component;
/**
* Normalizes mixed Chinese and ASCII search text into stable token sequences for
* both indexing and querying while keeping the raw phrase available for
* substring fallback matching.
*/
@Component
public class SearchTextTokenizer {
private static final Pattern WHITESPACE = Pattern.compile("\\s+");
private static final Pattern ASCII_TOKEN = Pattern.compile("[\\p{ASCII}]+");
private final JiebaSegmenter jiebaSegmenter = new JiebaSegmenter();
public List<String> tokenizeForIndex(String text) {
return tokenize(text, SegMode.INDEX);
}
public List<String> tokenizeForQuery(String text) {
return tokenize(text, SegMode.SEARCH);
}
public String enrichForIndex(String rawText) {
String normalized = normalize(rawText);
if (normalized == null) {
return "";
}
LinkedHashSet<String> parts = new LinkedHashSet<>();
parts.add(normalized);
parts.addAll(tokenizeForIndex(normalized));
return String.join(" ", parts);
}
private List<String> tokenize(String text, SegMode mode) {
String normalized = normalize(text);
if (normalized == null) {
return List.of();
}
Set<String> tokens = new LinkedHashSet<>();
for (SegToken token : jiebaSegmenter.process(normalized, mode)) {
String normalizedToken = normalizeToken(token.word);
if (normalizedToken != null) {
tokens.add(normalizedToken);
}
}
return List.copyOf(tokens);
}
private String normalize(String value) {
if (value == null) {
return null;
}
String normalized = WHITESPACE.matcher(value.trim()).replaceAll(" ");
return normalized.isBlank() ? null : normalized;
}
private String normalizeToken(String token) {
String normalized = normalize(token);
if (normalized == null) {
return null;
}
if (ASCII_TOKEN.matcher(normalized).matches()) {
normalized = normalized.toLowerCase(Locale.ROOT);
}
return normalized;
}
}

View file

@ -6,6 +6,7 @@ import com.iflytek.skillhub.search.SearchEmbeddingService;
import com.iflytek.skillhub.search.SearchQuery;
import com.iflytek.skillhub.search.SearchQueryService;
import com.iflytek.skillhub.search.SearchResult;
import com.iflytek.skillhub.search.SearchTextTokenizer;
import jakarta.persistence.EntityManager;
import jakarta.persistence.Query;
import java.util.Comparator;
@ -17,7 +18,6 @@ import org.springframework.beans.factory.annotation.Value;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.regex.Pattern;
/**
* PostgreSQL-backed implementation of {@link SearchQueryService}.
@ -28,7 +28,6 @@ import java.util.regex.Pattern;
*/
@Service
public class PostgresFullTextQueryService implements SearchQueryService {
private static final Pattern QUERY_TERM_SPLITTER = Pattern.compile("[^\\p{L}\\p{N}_]+");
private static final int MAX_QUERY_TERMS = 8;
private static final int SHORT_PREFIX_LENGTH = 2;
private static final String TITLE_VECTOR_SQL = "to_tsvector('simple', coalesce(title, ''))";
@ -37,19 +36,21 @@ public class PostgresFullTextQueryService implements SearchQueryService {
private final EntityManager entityManager;
private final SkillSearchDocumentJpaRepository searchDocumentRepository;
private final SearchEmbeddingService searchEmbeddingService;
private final SearchTextTokenizer searchTextTokenizer;
private final boolean semanticEnabled;
private final double semanticWeight;
private final int candidateMultiplier;
private final int maxCandidates;
public PostgresFullTextQueryService(EntityManager entityManager) {
this(entityManager, null, null, false, 0.35D, 8, 120);
this(entityManager, null, null, new SearchTextTokenizer(), false, 0.35D, 8, 120);
}
@Autowired
public PostgresFullTextQueryService(EntityManager entityManager,
SkillSearchDocumentJpaRepository searchDocumentRepository,
SearchEmbeddingService searchEmbeddingService,
SearchTextTokenizer searchTextTokenizer,
@Value("${skillhub.search.semantic.enabled:true}") boolean semanticEnabled,
@Value("${skillhub.search.semantic.weight:0.35}") double semanticWeight,
@Value("${skillhub.search.semantic.candidate-multiplier:8}") int candidateMultiplier,
@ -57,6 +58,7 @@ public class PostgresFullTextQueryService implements SearchQueryService {
this.entityManager = entityManager;
this.searchDocumentRepository = searchDocumentRepository;
this.searchEmbeddingService = searchEmbeddingService;
this.searchTextTokenizer = searchTextTokenizer;
this.semanticEnabled = semanticEnabled;
this.semanticWeight = semanticWeight;
this.candidateMultiplier = candidateMultiplier;
@ -74,7 +76,7 @@ public class PostgresFullTextQueryService implements SearchQueryService {
boolean hasKeyword = normalizedKeyword != null;
boolean hasTsQuery = tsQuery != null;
boolean useRelevanceOrdering = "relevance".equals(query.sortBy()) && hasKeyword;
boolean useShortPrefixTitleSearch = hasTsQuery && normalizedKeyword.length() <= SHORT_PREFIX_LENGTH;
boolean useShortPrefixTitleSearch = hasTsQuery && isShortAsciiPrefixSearch(normalizedKeyword);
boolean useSemanticRerank = semanticEnabled
&& hasKeyword
&& "relevance".equals(query.sortBy())
@ -299,10 +301,7 @@ public class PostgresFullTextQueryService implements SearchQueryService {
return null;
}
List<String> terms = QUERY_TERM_SPLITTER.splitAsStream(keyword.toLowerCase())
.map(String::trim)
.filter(term -> !term.isBlank())
.distinct()
List<String> terms = searchTextTokenizer.tokenizeForQuery(keyword).stream()
.limit(MAX_QUERY_TERMS)
.toList();
@ -319,13 +318,25 @@ public class PostgresFullTextQueryService implements SearchQueryService {
}
return tsQueryTerms.stream()
.map(term -> term + ":*")
.map(term -> usePrefixMatch(term) ? term + ":*" : term)
.reduce((left, right) -> left + " & " + right)
.orElse(null);
}
private boolean isTsQueryCompatibleTerm(String term) {
return term.chars().anyMatch(ch -> Character.isLetter(ch) || ch == '_');
return term.chars().anyMatch(ch -> Character.isLetter(ch) || Character.isIdeographic(ch) || ch == '_');
}
private boolean usePrefixMatch(String term) {
return term.chars().allMatch(ch -> ch < 128) && term.chars().anyMatch(Character::isLetter);
}
private boolean isShortAsciiPrefixSearch(String keyword) {
if (keyword == null || keyword.length() > SHORT_PREFIX_LENGTH) {
return false;
}
List<String> terms = searchTextTokenizer.tokenizeForQuery(keyword);
return !terms.isEmpty() && terms.stream().allMatch(this::usePrefixMatch);
}
private record RankedSkill(Long skillId, double score) {

View file

@ -11,6 +11,7 @@ import com.iflytek.skillhub.domain.skill.SkillVersion;
import com.iflytek.skillhub.domain.skill.SkillVersionRepository;
import com.iflytek.skillhub.search.SearchIndexService;
import com.iflytek.skillhub.search.SearchRebuildService;
import com.iflytek.skillhub.search.SearchTextTokenizer;
import com.iflytek.skillhub.search.SkillSearchDocument;
import org.springframework.stereotype.Service;
@ -37,17 +38,20 @@ public class PostgresSearchRebuildService implements SearchRebuildService {
private final NamespaceRepository namespaceRepository;
private final SkillVersionRepository skillVersionRepository;
private final SearchIndexService searchIndexService;
private final SearchTextTokenizer searchTextTokenizer;
private final ObjectMapper objectMapper;
public PostgresSearchRebuildService(
SkillRepository skillRepository,
NamespaceRepository namespaceRepository,
SkillVersionRepository skillVersionRepository,
SearchIndexService searchIndexService) {
SearchIndexService searchIndexService,
SearchTextTokenizer searchTextTokenizer) {
this.skillRepository = skillRepository;
this.namespaceRepository = namespaceRepository;
this.skillVersionRepository = skillVersionRepository;
this.searchIndexService = searchIndexService;
this.searchTextTokenizer = searchTextTokenizer;
this.objectMapper = new ObjectMapper();
}
@ -94,8 +98,8 @@ public class PostgresSearchRebuildService implements SearchRebuildService {
.ifPresent(frontmatter -> appendFrontmatter(frontmatter, keywords, searchParts));
return new SearchIndexPayload(
String.join(", ", keywords),
String.join(" ", searchParts).trim()
searchTextTokenizer.enrichForIndex(String.join(" ", keywords)),
searchTextTokenizer.enrichForIndex(String.join(" ", searchParts).trim())
);
}

View file

@ -0,0 +1,24 @@
package com.iflytek.skillhub.search;
import org.junit.jupiter.api.Test;
import static org.assertj.core.api.Assertions.assertThat;
class SearchTextTokenizerTest {
private final SearchTextTokenizer tokenizer = new SearchTextTokenizer();
@Test
void enrichForIndexShouldKeepRawPhraseAndAddSegmentedTokens() {
String enriched = tokenizer.enrichForIndex("中文技能搜索");
assertThat(enriched).contains("中文技能搜索");
assertThat(tokenizer.tokenizeForIndex("中文技能搜索")).contains("中文", "技能", "搜索");
}
@Test
void tokenizeForQueryShouldNormalizeAsciiTerms() {
assertThat(tokenizer.tokenizeForQuery("OpenAI Agent"))
.containsExactly("openai", "agent");
}
}

View file

@ -4,6 +4,7 @@ import com.iflytek.skillhub.infra.jpa.SkillSearchDocumentEntity;
import com.iflytek.skillhub.infra.jpa.SkillSearchDocumentJpaRepository;
import com.iflytek.skillhub.search.HashingSearchEmbeddingService;
import com.iflytek.skillhub.search.SearchQuery;
import com.iflytek.skillhub.search.SearchTextTokenizer;
import com.iflytek.skillhub.search.SearchVisibilityScope;
import jakarta.persistence.EntityManager;
import jakarta.persistence.Query;
@ -142,6 +143,36 @@ class PostgresFullTextQueryServiceTest {
assertThat(sqlCaptor.getAllValues().getFirst()).doesNotContain("ORDER BY ORDER BY");
}
@Test
void shortChineseKeywordsShouldStillUseSearchVectorInsteadOfTitlePrefixOptimization() {
EntityManager entityManager = mock(EntityManager.class);
Query nativeQuery = mock(Query.class);
Query countQuery = mock(Query.class);
when(entityManager.createNativeQuery(anyString()))
.thenReturn(nativeQuery)
.thenReturn(countQuery);
when(nativeQuery.setParameter(anyString(), org.mockito.ArgumentMatchers.any())).thenReturn(nativeQuery);
when(countQuery.setParameter(anyString(), org.mockito.ArgumentMatchers.any())).thenReturn(countQuery);
when(nativeQuery.getResultList()).thenReturn(List.of());
when(countQuery.getSingleResult()).thenReturn(0L);
PostgresFullTextQueryService service = new PostgresFullTextQueryService(entityManager);
service.search(new SearchQuery(
"中文",
null,
new SearchVisibilityScope(null, Set.of(), Set.of()),
"relevance",
0,
20
));
var sqlCaptor = org.mockito.ArgumentCaptor.forClass(String.class);
verify(entityManager, org.mockito.Mockito.times(2)).createNativeQuery(sqlCaptor.capture());
assertThat(sqlCaptor.getAllValues().getFirst()).contains("d.search_vector @@ to_tsquery('simple', :tsQuery)");
assertThat(sqlCaptor.getAllValues().getFirst()).doesNotContain("ts_rank_cd(to_tsvector('simple', coalesce(title, '')), to_tsquery('simple', :tsQuery))");
}
@Test
void multipleTermsShouldBuildPrefixQueryForEachLexeme() {
EntityManager entityManager = mock(EntityManager.class);
@ -170,6 +201,34 @@ class PostgresFullTextQueryServiceTest {
verify(countQuery).setParameter("tsQuery", "self:* & improving:*");
}
@Test
void chineseKeywordsShouldUseSegmentedTermsWithoutAsciiPrefixSuffix() {
EntityManager entityManager = mock(EntityManager.class);
Query nativeQuery = mock(Query.class);
Query countQuery = mock(Query.class);
when(entityManager.createNativeQuery(anyString()))
.thenReturn(nativeQuery)
.thenReturn(countQuery);
when(nativeQuery.setParameter(anyString(), org.mockito.ArgumentMatchers.any())).thenReturn(nativeQuery);
when(countQuery.setParameter(anyString(), org.mockito.ArgumentMatchers.any())).thenReturn(countQuery);
when(nativeQuery.getResultList()).thenReturn(List.of());
when(countQuery.getSingleResult()).thenReturn(0L);
PostgresFullTextQueryService service = new PostgresFullTextQueryService(entityManager);
service.search(new SearchQuery(
"中文技能搜索",
null,
new SearchVisibilityScope(null, Set.of(), Set.of()),
"relevance",
0,
20
));
verify(nativeQuery).setParameter("tsQuery", "中文 & 技能 & 搜索");
verify(countQuery).setParameter("tsQuery", "中文 & 技能 & 搜索");
}
@Test
void numericKeywordsShouldFallbackToTitleLikeSearchWithoutTsQuery() {
EntityManager entityManager = mock(EntityManager.class);
@ -391,6 +450,7 @@ class PostgresFullTextQueryServiceTest {
entityManager,
repository,
embeddingService,
new SearchTextTokenizer(),
true,
0.6D,
8,
@ -430,6 +490,7 @@ class PostgresFullTextQueryServiceTest {
entityManager,
repository,
embeddingService,
new SearchTextTokenizer(),
true,
0.6D,
8,

View file

@ -8,6 +8,7 @@ import com.iflytek.skillhub.domain.skill.SkillVersion;
import com.iflytek.skillhub.domain.skill.SkillVersionRepository;
import com.iflytek.skillhub.domain.skill.SkillVisibility;
import com.iflytek.skillhub.search.SearchIndexService;
import com.iflytek.skillhub.search.SearchTextTokenizer;
import com.iflytek.skillhub.search.SkillSearchDocument;
import org.junit.jupiter.api.Test;
import org.mockito.ArgumentCaptor;
@ -64,7 +65,8 @@ class PostgresSearchRebuildServiceTest {
skillRepository,
namespaceRepository,
skillVersionRepository,
searchIndexService
searchIndexService,
new SearchTextTokenizer()
);
service.rebuildBySkill(1L);
@ -73,7 +75,12 @@ class PostgresSearchRebuildServiceTest {
verify(searchIndexService).index(captor.capture());
SkillSearchDocument document = captor.getValue();
assertThat(document.keywords()).isEqualTo("agentic, assistant, automation, workflow");
assertThat(document.title()).isEqualTo("Smart Agent");
assertThat(document.summary()).isEqualTo("Builds workflows");
assertThat(document.keywords()).contains("agentic");
assertThat(document.keywords()).contains("assistant");
assertThat(document.keywords()).contains("automation");
assertThat(document.keywords()).contains("workflow");
assertThat(document.searchText()).contains("Smart Agent");
assertThat(document.searchText()).contains("smart-agent");
assertThat(document.searchText()).contains("Builds workflows");