feat: Optimized FlexSearch document and encoder settings to improve autocompletion

This commit is contained in:
newtextdoc1111
2025-07-23 07:47:52 +09:00
parent 083c1f5acc
commit 6c29336435
4 changed files with 169 additions and 129 deletions
+47 -64
View File
@@ -1,5 +1,10 @@
import { Document, Charset, Encoder } from '../../web/js/thirdparty/flexsearch.bundle.module.min.js'
import {
createFlexSearchDocument,
__test__
} from "../../web/js/searchengine.js";
const { createTagEncoder, createCJKEncoder } = __test__;
function parseCSVLine(line) {
const result = [];
@@ -36,6 +41,7 @@ describe('FlexSearch Integration', () => {
highres,5,5256195,"high_res,high_resolution,hires"
solo,0,5000954,"alone,female_solo,single,solo_female,solo_in_panel"
long_hair,0,4350743,"/lh,longhair,very_long_hair"
one_two_three,0,29389,
`;
const cjkAliasCSV = `
@@ -62,18 +68,21 @@ __wildcard__,0,1000,
Embedding: my_embedding,0,1000,
`;
const mockCSV = [commonCSV, cjkAliasCSV, specialCharCSV, ControlCSV].map(csv => csv.trim()).join('\n');
let mockTags = [];
const mockCSV = [
commonCSV, cjkAliasCSV, specialCharCSV, ControlCSV
].map(csv => csv.trim()).join('\n');
let mockTags;
let tagEncoder, cjkEncoder, customEncoder
let index;
let tagEncoder, cjkEncoder;
let document;
let performSearch = function (query, limit = null) {
const results = index.search(query, { field: ["tag", "alias"], limit: limit, suggest: false });
const results = document.search(query, { field: ["tag", "alias"], limit: limit, suggest: false });
const ids = results.map(r => r.result).flat();
return mockTags.filter(tag => ids.includes(tag.id)).map(tag => tag.tag);;
return mockTags.filter(tag => ids.includes(tag.id)).map(tag => tag.tag);
}
beforeEach(() => {
@@ -82,84 +91,44 @@ Embedding: my_embedding,0,1000,
return { id, tag, category: parseInt(category), count: parseInt(count), alias };
});
tagEncoder = new Encoder({
normalize: true,
dedupe: false,
numeric: false,
cache: true,
split: /(?<=[a-zA-Z\)])_(?=[a-zA-Z\(])|\((?=[a-zA-Z])|(?<=[a-zA-Z\)])\)|[ \n]/
});
tagEncoder = createTagEncoder();
cjkEncoder = createCJKEncoder();
cjkEncoder = new Encoder(Charset.CJK, {
dedupe: true,
numeric: true,
cache: false,
filter: new Set(['_', '(', ')']),
finalize: (term) => {
return term.map(str => str.replace(/[\u30a1-\u30f6]/g, function (match) {
const chr = match.charCodeAt(0) - 0x60;
return String.fromCharCode(chr);
}));
}
});
document = createFlexSearchDocument();
customEncoder = function (term) {
return term.split(",")
.flatMap(str => {
if (/[^\u0000-\u007f]/.test(str)) {
// Contains non-ASCII characters
return cjkEncoder.encode(str);
} else {
// ASCII characters only
return tagEncoder.encode(str);
}
})
.filter(Boolean);
}
index = new Document({
document: {
id: "id",
index: [
{
field: "tag",
tokenize: "bidirectional",
encoder: tagEncoder,
},
{
field: "alias",
tokenize: "default",
encode: customEncoder,
}
]
}
});
mockTags.forEach(data => index.add(data));
mockTags.forEach(data => document.add(data));
});
describe('Encoder', () => {
test('should be encoded 1', () => {
test('should split underscore-separated tags', () => {
const encoded = tagEncoder.encode('sanshoku_dango');
expect(encoded).toEqual(['sanshoku', 'dango']);
});
test('should be encoded 2', () => {
test('should extract words from parentheses', () => {
const encoded = tagEncoder.encode('copyright_(series)');
expect(encoded).toEqual(['copyright', 'series']);
});
test('should be encoded 3', () => {
test('should preserve colon-separated special tags', () => {
const encoded = tagEncoder.encode('year:1234');
expect(encoded).toEqual(['year:1234']);
});
test('should be encoded 4', () => {
test('should preserve dot-separated tags', () => {
const encoded = tagEncoder.encode('d.d.');
expect(encoded).toEqual(['d.d.']);
});
test('should be encoded 5', () => {
test('should preserve double underscore wildcard tags', () => {
const encoded = tagEncoder.encode('__wildcard__');
expect(encoded).toEqual(['__wildcard__']);
});
test('should convert katakana to hiragana', () => {
const encoded = cjkEncoder.encode('ガーデン');
expect(encoded).toEqual(['がーでん']);
});
test('should remove trailing underscores when splitting', () => {
const encoded = tagEncoder.encode('one_two_');
expect(encoded).toEqual(['one', 'two']);
});
});
describe('Basic Search', () => {
@@ -188,6 +157,20 @@ Embedding: my_embedding,0,1000,
expect(results).toContain('dragon_girl');
});
test('should find a tag for terms contain space', () => {
const results = performSearch('double ');
expect(results.length).toEqual(1);
expect(results).toContain('double_bun');
});
test('should find a tag for terms contain underscore', () => {
const results = performSearch('double_');
expect(results.length).toEqual(1);
expect(results).toContain('double_bun');
});
});
describe('Alias Search', () => {
+28 -37
View File
@@ -83,8 +83,8 @@ function searchCompletionCandidates(textareaElement) {
const ESCAPE_SEQUENCE = ["#", "/"]; // If the first string is that character, autocomplete will not be displayed.
const partialTag = getCurrentPartialTag(textareaElement);
if (!partialTag || partialTag.length <= 0 ||
ESCAPE_SEQUENCE.some(seq => partialTag.startsWith(seq)) ||
if (!partialTag || partialTag.length <= 0 ||
ESCAPE_SEQUENCE.some(seq => partialTag.startsWith(seq)) ||
isLongText(partialTag)) {
return []; // No valid input for autocomplete
}
@@ -107,50 +107,41 @@ function searchCompletionCandidates(textareaElement) {
const sources = getEnabledTagSourceInPriorityOrder();
for (const source of sources) {
// Use fast search if enabled and available for the source
if (settingValues.useFastSearch && autoCompleteData[source].flexSearchIndex) {
// Use the FlexSearch Index to search tag and alias IDs that match the partial tag.
const searchResults = autoCompleteData[source].flexSearchIndex.search(partialTag, {
limit: settingValues.maxSuggestions * autoCompleteData[source].flexSearchLimitMultiplier,
if (settingValues.useFastSearch && autoCompleteData[source].flexSearchDocument) {
// Use the FlexSearch Document to search
let result = autoCompleteData[source].flexSearchDocument.search(partialTag, {
field: ["tag", "alias"],
limit: settingValues.maxSuggestions,
merge: true,
suggest: false,
cache: true,
});
// Get tag IDs from search results and filter duplicates
let result = searchResults.map((index) => {
return autoCompleteData[source].flexSearchMapping[index];
});
result = [...new Set(result)];
// Sort results based on exact matches or id values (ID order is equal to tag count order)
result = result.sort((a, b) => {
const aTag = autoCompleteData[source].sortedTags[a];
const bTag = autoCompleteData[source].sortedTags[b];
if (matchWord(bTag.tag, queryVariations).isExactMatch) {
return 999999999999;
}
if (matchWord(aTag.tag, queryVariations).isExactMatch) {
return -999999999999;
}
if (bTag.alias && bTag.alias.some(alias => matchWord(alias, queryVariations).isExactMatch)) {
return 999999999999;
}
if (aTag.alias && aTag.alias.some(alias => matchWord(alias, queryVariations).isExactMatch)) {
return -999999999999;
}
return a - b;
});
// Limit the results to maxSuggestions and map to TagData
result = result.slice(0, Math.min(result.length, settingValues.maxSuggestions));
result = result.map((index) => {
return autoCompleteData[source].sortedTags[index];
});
// Sort results based on exact matches or tag count
result = result
.map(r => autoCompleteData[source].sortedTags[r.id])
.sort((aTag, bTag) => {
if (matchWord(bTag.tag, queryVariations).isExactMatch) {
return 999999999999;
}
if (matchWord(aTag.tag, queryVariations).isExactMatch) {
return -999999999999;
}
if (bTag.alias && bTag.alias.some(alias => matchWord(alias, queryVariations).isExactMatch)) {
return 999999999999;
}
if (aTag.alias && aTag.alias.some(alias => matchWord(alias, queryVariations).isExactMatch)) {
return -999999999999;
}
return bTag.count - aTag.count;
});
if (settingValues._logprocessingTime) {
const endTime = performance.now();
const duration = endTime - startTime;
console.debug(`[Autocomplete-Plus] Fast Search for "${partialTag}" in ${source} took ${duration.toFixed(2)}ms. Found ${result.length} candidates within ${searchResults.length} searches with aliases.`);
console.debug(`[Autocomplete-Plus] Fast Search for "${partialTag}" in ${source} took ${duration.toFixed(2)}ms. Found ${result.length} candidates.`);
}
return result;
} else {
// Search in sortedTags (already sorted by count)
+8 -28
View File
@@ -1,5 +1,5 @@
import { Index } from './thirdparty/flexsearch.bundle.module.min.js'
import { settingValues, updateMaxTagLength } from "./settings.js";
import { createFlexSearchDocument } from "./searchengine.js";
// --- Constants ---
@@ -64,15 +64,8 @@ export class TagData {
class AutocompleteData {
constructor() {
/** @type {Index} */
this.flexSearchIndex = null;
/** @type {number[]} */
this.flexSearchMapping = [];
/** @type {number} */
// The actual number will be calculated later when loading CSV files
this.flexSearchLimitMultiplier = 10;
/** @type {Document} */
this.flexSearchDocument = null;
/** @type {TagData[]} */
this.sortedTags = [];
@@ -91,7 +84,6 @@ class AutocompleteData {
// Progress of "base" csv loading
this.baseLoadingProgress = {
// tags: 0,
cooccurrence: 0
};
}
@@ -210,39 +202,27 @@ async function buildFlexSearchIndex(siteName) {
return;
}
const index = new Index({
tokenize: "bidirectional",
});
const document = createFlexSearchDocument();
let startIdx = 0;
let maxCountOfAlias = 0;
const startTime = performance.now();
function processChunkTasks() {
const chunkSize = 1000;
const end = Math.min(startIdx + chunkSize, autoCompleteData[siteName].sortedTags.length);
for (; startIdx < end; startIdx++) {
const tagData = autoCompleteData[siteName].sortedTags[startIdx];
index.add(autoCompleteData[siteName].flexSearchMapping.length, tagData.tag);
autoCompleteData[siteName].flexSearchMapping.push(startIdx);
tagData.alias.forEach(alias => {
index.add(autoCompleteData[siteName].flexSearchMapping.length, alias);
autoCompleteData[siteName].flexSearchMapping.push(startIdx);
})
maxCountOfAlias = Math.max(maxCountOfAlias, tagData.alias.length);
document.add(startIdx, tagData);
}
if (startIdx < autoCompleteData[siteName].sortedTags.length) {
setTimeout(processChunkTasks, 0);
// console.log(`[Autocomplete-Plus] Current porcess: ${startIdx}`);
} else {
autoCompleteData[siteName].flexSearchDocument = document;
const endTime = performance.now();
const duration = endTime - startTime;
autoCompleteData[siteName].flexSearchIndex = index;
autoCompleteData[siteName].flexSearchLimitMultiplier = Math.min(10, maxCountOfAlias + 1);
console.debug(`[Autocomplete-Plus] Building ${autoCompleteData[siteName].sortedTags.length} index for ${siteName} took ${duration.toFixed(2)}ms.`);
console.info(`[Autocomplete-Plus] Building ${autoCompleteData[siteName].sortedTags.length} index for ${siteName} took ${duration.toFixed(2)}ms.`);
}
}
processChunkTasks();
+86
View File
@@ -0,0 +1,86 @@
import { Charset, Encoder, Document } from './thirdparty/flexsearch.bundle.module.min.js'
import { kataToHira } from './utils.js';
/**
* Creates an encoder optimized for processing English tag names.
* Handles tag formatting like underscores and parentheses commonly used in Danbooru tags.
* @returns {Encoder} FlexSearch encoder for English tags
*/
function createTagEncoder() {
return new Encoder({
normalize: true,
dedupe: false,
numeric: false,
cache: true,
// filter: new Set(['and', 'to', 'be', 'on']),
replacer: [/(?<=[a-zA-Z\)])_$/, ''], // Remove trailing underscores after letters/parentheses
split: /(?<=[a-zA-Z\)])_(?=[a-zA-Z\(])|\((?=[a-zA-Z])|(?<=[a-zA-Z\)])\)|[ \n]/ // Split on underscores between words, parentheses, spaces, and newlines
});
}
/**
* Creates an encoder optimized for processing CJK (Chinese, Japanese, Korean) characters.
* Uses exact character matching and converts katakana to hiragana for better Japanese search.
* @returns {Encoder} FlexSearch encoder for CJK text
*/
function createCJKEncoder() {
return new Encoder(Charset.Exact, {
dedupe: true,
numeric: true,
cache: true,
filter: new Set(['(', ')']), // Filter out parentheses characters
finalize: (term) => { // Convert katakana to hiragana for better Japanese matching
return term.map(str => kataToHira(str));
}
});
}
/**
* Creates a FlexSearch Document instance optimized for tag searching.
* Configures separate encoders for English tags and CJK aliases with appropriate tokenization.
* @returns {Document} Configured FlexSearch document for tag indexing
*/
export function createFlexSearchDocument() {
const tagEncoder = createTagEncoder();
const cjkEncoder = createCJKEncoder();
// Custom encoding function for alias field that handles mixed language content
const encodeAlias = function (term) {
return term.split(",")
.flatMap(str => {
if (/[^\u0000-\u007f]/.test(str)) {
// Contains non-ASCII characters (CJK text)
return cjkEncoder.encode(str);
} else {
// ASCII characters only (English text)
return tagEncoder.encode(str);
}
})
.filter(Boolean);
}
// Configure the FlexSearch document with optimized indexing settings
const document = new Document({
document: {
id: "id",
index: [
{
field: "tag",
tokenize: "bidirectional", // Allow partial matching from both ends
encoder: tagEncoder, // Use tag-optimized encoder
},
{
field: "alias", // Index the alias field for multi-language support
tokenize: "full", // Full tokenization for complete alias matching
encode: encodeAlias, // Use custom multi-language encoding function
}
]
}
});
return document;
}
// Export functions for testing when in test environment
const isTestEnvironment = typeof process !== 'undefined' && process.env.NODE_ENV === 'test';
export const __test__ = isTestEnvironment ? { createTagEncoder, createCJKEncoder } : undefined;