feat: Optimized FlexSearch document and encoder settings to improve autocompletion
This commit is contained in:
+47
-64
@@ -1,5 +1,10 @@
|
||||
|
||||
import { Document, Charset, Encoder } from '../../web/js/thirdparty/flexsearch.bundle.module.min.js'
|
||||
import {
|
||||
createFlexSearchDocument,
|
||||
__test__
|
||||
} from "../../web/js/searchengine.js";
|
||||
|
||||
const { createTagEncoder, createCJKEncoder } = __test__;
|
||||
|
||||
function parseCSVLine(line) {
|
||||
const result = [];
|
||||
@@ -36,6 +41,7 @@ describe('FlexSearch Integration', () => {
|
||||
highres,5,5256195,"high_res,high_resolution,hires"
|
||||
solo,0,5000954,"alone,female_solo,single,solo_female,solo_in_panel"
|
||||
long_hair,0,4350743,"/lh,longhair,very_long_hair"
|
||||
one_two_three,0,29389,
|
||||
`;
|
||||
|
||||
const cjkAliasCSV = `
|
||||
@@ -62,18 +68,21 @@ __wildcard__,0,1000,
|
||||
Embedding: my_embedding,0,1000,
|
||||
`;
|
||||
|
||||
const mockCSV = [commonCSV, cjkAliasCSV, specialCharCSV, ControlCSV].map(csv => csv.trim()).join('\n');
|
||||
let mockTags = [];
|
||||
const mockCSV = [
|
||||
commonCSV, cjkAliasCSV, specialCharCSV, ControlCSV
|
||||
].map(csv => csv.trim()).join('\n');
|
||||
|
||||
let mockTags;
|
||||
|
||||
let tagEncoder, cjkEncoder, customEncoder
|
||||
let index;
|
||||
let tagEncoder, cjkEncoder;
|
||||
let document;
|
||||
|
||||
let performSearch = function (query, limit = null) {
|
||||
const results = index.search(query, { field: ["tag", "alias"], limit: limit, suggest: false });
|
||||
const results = document.search(query, { field: ["tag", "alias"], limit: limit, suggest: false });
|
||||
|
||||
const ids = results.map(r => r.result).flat();
|
||||
|
||||
return mockTags.filter(tag => ids.includes(tag.id)).map(tag => tag.tag);;
|
||||
return mockTags.filter(tag => ids.includes(tag.id)).map(tag => tag.tag);
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
@@ -82,84 +91,44 @@ Embedding: my_embedding,0,1000,
|
||||
return { id, tag, category: parseInt(category), count: parseInt(count), alias };
|
||||
});
|
||||
|
||||
tagEncoder = new Encoder({
|
||||
normalize: true,
|
||||
dedupe: false,
|
||||
numeric: false,
|
||||
cache: true,
|
||||
split: /(?<=[a-zA-Z\)])_(?=[a-zA-Z\(])|\((?=[a-zA-Z])|(?<=[a-zA-Z\)])\)|[ \n]/
|
||||
});
|
||||
tagEncoder = createTagEncoder();
|
||||
cjkEncoder = createCJKEncoder();
|
||||
|
||||
cjkEncoder = new Encoder(Charset.CJK, {
|
||||
dedupe: true,
|
||||
numeric: true,
|
||||
cache: false,
|
||||
filter: new Set(['_', '(', ')']),
|
||||
finalize: (term) => {
|
||||
return term.map(str => str.replace(/[\u30a1-\u30f6]/g, function (match) {
|
||||
const chr = match.charCodeAt(0) - 0x60;
|
||||
return String.fromCharCode(chr);
|
||||
}));
|
||||
}
|
||||
});
|
||||
document = createFlexSearchDocument();
|
||||
|
||||
customEncoder = function (term) {
|
||||
return term.split(",")
|
||||
.flatMap(str => {
|
||||
if (/[^\u0000-\u007f]/.test(str)) {
|
||||
// Contains non-ASCII characters
|
||||
return cjkEncoder.encode(str);
|
||||
} else {
|
||||
// ASCII characters only
|
||||
return tagEncoder.encode(str);
|
||||
}
|
||||
})
|
||||
.filter(Boolean);
|
||||
}
|
||||
|
||||
index = new Document({
|
||||
document: {
|
||||
id: "id",
|
||||
index: [
|
||||
{
|
||||
field: "tag",
|
||||
tokenize: "bidirectional",
|
||||
encoder: tagEncoder,
|
||||
},
|
||||
{
|
||||
field: "alias",
|
||||
tokenize: "default",
|
||||
encode: customEncoder,
|
||||
}
|
||||
]
|
||||
}
|
||||
});
|
||||
|
||||
mockTags.forEach(data => index.add(data));
|
||||
mockTags.forEach(data => document.add(data));
|
||||
});
|
||||
|
||||
describe('Encoder', () => {
|
||||
test('should be encoded 1', () => {
|
||||
test('should split underscore-separated tags', () => {
|
||||
const encoded = tagEncoder.encode('sanshoku_dango');
|
||||
expect(encoded).toEqual(['sanshoku', 'dango']);
|
||||
});
|
||||
|
||||
test('should be encoded 2', () => {
|
||||
test('should extract words from parentheses', () => {
|
||||
const encoded = tagEncoder.encode('copyright_(series)');
|
||||
expect(encoded).toEqual(['copyright', 'series']);
|
||||
});
|
||||
test('should be encoded 3', () => {
|
||||
test('should preserve colon-separated special tags', () => {
|
||||
const encoded = tagEncoder.encode('year:1234');
|
||||
expect(encoded).toEqual(['year:1234']);
|
||||
});
|
||||
test('should be encoded 4', () => {
|
||||
test('should preserve dot-separated tags', () => {
|
||||
const encoded = tagEncoder.encode('d.d.');
|
||||
expect(encoded).toEqual(['d.d.']);
|
||||
});
|
||||
test('should be encoded 5', () => {
|
||||
test('should preserve double underscore wildcard tags', () => {
|
||||
const encoded = tagEncoder.encode('__wildcard__');
|
||||
expect(encoded).toEqual(['__wildcard__']);
|
||||
});
|
||||
test('should convert katakana to hiragana', () => {
|
||||
const encoded = cjkEncoder.encode('ガーデン');
|
||||
expect(encoded).toEqual(['がーでん']);
|
||||
});
|
||||
test('should remove trailing underscores when splitting', () => {
|
||||
const encoded = tagEncoder.encode('one_two_');
|
||||
expect(encoded).toEqual(['one', 'two']);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Basic Search', () => {
|
||||
@@ -188,6 +157,20 @@ Embedding: my_embedding,0,1000,
|
||||
|
||||
expect(results).toContain('dragon_girl');
|
||||
});
|
||||
|
||||
test('should find a tag for terms contain space', () => {
|
||||
const results = performSearch('double ');
|
||||
expect(results.length).toEqual(1);
|
||||
|
||||
expect(results).toContain('double_bun');
|
||||
});
|
||||
|
||||
test('should find a tag for terms contain underscore', () => {
|
||||
const results = performSearch('double_');
|
||||
expect(results.length).toEqual(1);
|
||||
|
||||
expect(results).toContain('double_bun');
|
||||
});
|
||||
});
|
||||
|
||||
describe('Alias Search', () => {
|
||||
|
||||
+28
-37
@@ -83,8 +83,8 @@ function searchCompletionCandidates(textareaElement) {
|
||||
|
||||
const ESCAPE_SEQUENCE = ["#", "/"]; // If the first string is that character, autocomplete will not be displayed.
|
||||
const partialTag = getCurrentPartialTag(textareaElement);
|
||||
if (!partialTag || partialTag.length <= 0 ||
|
||||
ESCAPE_SEQUENCE.some(seq => partialTag.startsWith(seq)) ||
|
||||
if (!partialTag || partialTag.length <= 0 ||
|
||||
ESCAPE_SEQUENCE.some(seq => partialTag.startsWith(seq)) ||
|
||||
isLongText(partialTag)) {
|
||||
return []; // No valid input for autocomplete
|
||||
}
|
||||
@@ -107,50 +107,41 @@ function searchCompletionCandidates(textareaElement) {
|
||||
const sources = getEnabledTagSourceInPriorityOrder();
|
||||
for (const source of sources) {
|
||||
// Use fast search if enabled and available for the source
|
||||
if (settingValues.useFastSearch && autoCompleteData[source].flexSearchIndex) {
|
||||
// Use the FlexSearch Index to search tag and alias IDs that match the partial tag.
|
||||
const searchResults = autoCompleteData[source].flexSearchIndex.search(partialTag, {
|
||||
limit: settingValues.maxSuggestions * autoCompleteData[source].flexSearchLimitMultiplier,
|
||||
if (settingValues.useFastSearch && autoCompleteData[source].flexSearchDocument) {
|
||||
// Use the FlexSearch Document to search
|
||||
let result = autoCompleteData[source].flexSearchDocument.search(partialTag, {
|
||||
field: ["tag", "alias"],
|
||||
limit: settingValues.maxSuggestions,
|
||||
merge: true,
|
||||
suggest: false,
|
||||
cache: true,
|
||||
});
|
||||
|
||||
// Get tag IDs from search results and filter duplicates
|
||||
let result = searchResults.map((index) => {
|
||||
return autoCompleteData[source].flexSearchMapping[index];
|
||||
});
|
||||
result = [...new Set(result)];
|
||||
|
||||
// Sort results based on exact matches or id values (ID order is equal to tag count order)
|
||||
result = result.sort((a, b) => {
|
||||
const aTag = autoCompleteData[source].sortedTags[a];
|
||||
const bTag = autoCompleteData[source].sortedTags[b];
|
||||
if (matchWord(bTag.tag, queryVariations).isExactMatch) {
|
||||
return 999999999999;
|
||||
}
|
||||
if (matchWord(aTag.tag, queryVariations).isExactMatch) {
|
||||
return -999999999999;
|
||||
}
|
||||
if (bTag.alias && bTag.alias.some(alias => matchWord(alias, queryVariations).isExactMatch)) {
|
||||
return 999999999999;
|
||||
}
|
||||
if (aTag.alias && aTag.alias.some(alias => matchWord(alias, queryVariations).isExactMatch)) {
|
||||
return -999999999999;
|
||||
}
|
||||
return a - b;
|
||||
});
|
||||
|
||||
// Limit the results to maxSuggestions and map to TagData
|
||||
result = result.slice(0, Math.min(result.length, settingValues.maxSuggestions));
|
||||
result = result.map((index) => {
|
||||
return autoCompleteData[source].sortedTags[index];
|
||||
});
|
||||
// Sort results based on exact matches or tag count
|
||||
result = result
|
||||
.map(r => autoCompleteData[source].sortedTags[r.id])
|
||||
.sort((aTag, bTag) => {
|
||||
if (matchWord(bTag.tag, queryVariations).isExactMatch) {
|
||||
return 999999999999;
|
||||
}
|
||||
if (matchWord(aTag.tag, queryVariations).isExactMatch) {
|
||||
return -999999999999;
|
||||
}
|
||||
if (bTag.alias && bTag.alias.some(alias => matchWord(alias, queryVariations).isExactMatch)) {
|
||||
return 999999999999;
|
||||
}
|
||||
if (aTag.alias && aTag.alias.some(alias => matchWord(alias, queryVariations).isExactMatch)) {
|
||||
return -999999999999;
|
||||
}
|
||||
return bTag.count - aTag.count;
|
||||
});
|
||||
|
||||
if (settingValues._logprocessingTime) {
|
||||
const endTime = performance.now();
|
||||
const duration = endTime - startTime;
|
||||
console.debug(`[Autocomplete-Plus] Fast Search for "${partialTag}" in ${source} took ${duration.toFixed(2)}ms. Found ${result.length} candidates within ${searchResults.length} searches with aliases.`);
|
||||
console.debug(`[Autocomplete-Plus] Fast Search for "${partialTag}" in ${source} took ${duration.toFixed(2)}ms. Found ${result.length} candidates.`);
|
||||
}
|
||||
|
||||
return result;
|
||||
} else {
|
||||
// Search in sortedTags (already sorted by count)
|
||||
|
||||
+8
-28
@@ -1,5 +1,5 @@
|
||||
import { Index } from './thirdparty/flexsearch.bundle.module.min.js'
|
||||
import { settingValues, updateMaxTagLength } from "./settings.js";
|
||||
import { createFlexSearchDocument } from "./searchengine.js";
|
||||
|
||||
// --- Constants ---
|
||||
|
||||
@@ -64,15 +64,8 @@ export class TagData {
|
||||
|
||||
class AutocompleteData {
|
||||
constructor() {
|
||||
/** @type {Index} */
|
||||
this.flexSearchIndex = null;
|
||||
|
||||
/** @type {number[]} */
|
||||
this.flexSearchMapping = [];
|
||||
|
||||
/** @type {number} */
|
||||
// The actual number will be calculated later when loading CSV files
|
||||
this.flexSearchLimitMultiplier = 10;
|
||||
/** @type {Document} */
|
||||
this.flexSearchDocument = null;
|
||||
|
||||
/** @type {TagData[]} */
|
||||
this.sortedTags = [];
|
||||
@@ -91,7 +84,6 @@ class AutocompleteData {
|
||||
|
||||
// Progress of "base" csv loading
|
||||
this.baseLoadingProgress = {
|
||||
// tags: 0,
|
||||
cooccurrence: 0
|
||||
};
|
||||
}
|
||||
@@ -210,39 +202,27 @@ async function buildFlexSearchIndex(siteName) {
|
||||
return;
|
||||
}
|
||||
|
||||
const index = new Index({
|
||||
tokenize: "bidirectional",
|
||||
});
|
||||
const document = createFlexSearchDocument();
|
||||
|
||||
let startIdx = 0;
|
||||
let maxCountOfAlias = 0;
|
||||
const startTime = performance.now();
|
||||
function processChunkTasks() {
|
||||
const chunkSize = 1000;
|
||||
const end = Math.min(startIdx + chunkSize, autoCompleteData[siteName].sortedTags.length);
|
||||
for (; startIdx < end; startIdx++) {
|
||||
const tagData = autoCompleteData[siteName].sortedTags[startIdx];
|
||||
|
||||
index.add(autoCompleteData[siteName].flexSearchMapping.length, tagData.tag);
|
||||
autoCompleteData[siteName].flexSearchMapping.push(startIdx);
|
||||
|
||||
tagData.alias.forEach(alias => {
|
||||
index.add(autoCompleteData[siteName].flexSearchMapping.length, alias);
|
||||
autoCompleteData[siteName].flexSearchMapping.push(startIdx);
|
||||
})
|
||||
|
||||
maxCountOfAlias = Math.max(maxCountOfAlias, tagData.alias.length);
|
||||
document.add(startIdx, tagData);
|
||||
}
|
||||
|
||||
if (startIdx < autoCompleteData[siteName].sortedTags.length) {
|
||||
setTimeout(processChunkTasks, 0);
|
||||
// console.log(`[Autocomplete-Plus] Current porcess: ${startIdx}`);
|
||||
} else {
|
||||
autoCompleteData[siteName].flexSearchDocument = document;
|
||||
|
||||
const endTime = performance.now();
|
||||
const duration = endTime - startTime;
|
||||
autoCompleteData[siteName].flexSearchIndex = index;
|
||||
autoCompleteData[siteName].flexSearchLimitMultiplier = Math.min(10, maxCountOfAlias + 1);
|
||||
console.debug(`[Autocomplete-Plus] Building ${autoCompleteData[siteName].sortedTags.length} index for ${siteName} took ${duration.toFixed(2)}ms.`);
|
||||
console.info(`[Autocomplete-Plus] Building ${autoCompleteData[siteName].sortedTags.length} index for ${siteName} took ${duration.toFixed(2)}ms.`);
|
||||
}
|
||||
}
|
||||
processChunkTasks();
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
import { Charset, Encoder, Document } from './thirdparty/flexsearch.bundle.module.min.js'
|
||||
import { kataToHira } from './utils.js';
|
||||
|
||||
/**
|
||||
* Creates an encoder optimized for processing English tag names.
|
||||
* Handles tag formatting like underscores and parentheses commonly used in Danbooru tags.
|
||||
* @returns {Encoder} FlexSearch encoder for English tags
|
||||
*/
|
||||
function createTagEncoder() {
|
||||
return new Encoder({
|
||||
normalize: true,
|
||||
dedupe: false,
|
||||
numeric: false,
|
||||
cache: true,
|
||||
// filter: new Set(['and', 'to', 'be', 'on']),
|
||||
replacer: [/(?<=[a-zA-Z\)])_$/, ''], // Remove trailing underscores after letters/parentheses
|
||||
split: /(?<=[a-zA-Z\)])_(?=[a-zA-Z\(])|\((?=[a-zA-Z])|(?<=[a-zA-Z\)])\)|[ \n]/ // Split on underscores between words, parentheses, spaces, and newlines
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates an encoder optimized for processing CJK (Chinese, Japanese, Korean) characters.
|
||||
* Uses exact character matching and converts katakana to hiragana for better Japanese search.
|
||||
* @returns {Encoder} FlexSearch encoder for CJK text
|
||||
*/
|
||||
function createCJKEncoder() {
|
||||
return new Encoder(Charset.Exact, {
|
||||
dedupe: true,
|
||||
numeric: true,
|
||||
cache: true,
|
||||
filter: new Set(['(', ')']), // Filter out parentheses characters
|
||||
finalize: (term) => { // Convert katakana to hiragana for better Japanese matching
|
||||
return term.map(str => kataToHira(str));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Creates a FlexSearch Document instance optimized for tag searching.
|
||||
* Configures separate encoders for English tags and CJK aliases with appropriate tokenization.
|
||||
* @returns {Document} Configured FlexSearch document for tag indexing
|
||||
*/
|
||||
export function createFlexSearchDocument() {
|
||||
const tagEncoder = createTagEncoder();
|
||||
const cjkEncoder = createCJKEncoder();
|
||||
|
||||
// Custom encoding function for alias field that handles mixed language content
|
||||
const encodeAlias = function (term) {
|
||||
return term.split(",")
|
||||
.flatMap(str => {
|
||||
if (/[^\u0000-\u007f]/.test(str)) {
|
||||
// Contains non-ASCII characters (CJK text)
|
||||
return cjkEncoder.encode(str);
|
||||
} else {
|
||||
// ASCII characters only (English text)
|
||||
return tagEncoder.encode(str);
|
||||
}
|
||||
})
|
||||
.filter(Boolean);
|
||||
}
|
||||
|
||||
// Configure the FlexSearch document with optimized indexing settings
|
||||
const document = new Document({
|
||||
document: {
|
||||
id: "id",
|
||||
index: [
|
||||
{
|
||||
field: "tag",
|
||||
tokenize: "bidirectional", // Allow partial matching from both ends
|
||||
encoder: tagEncoder, // Use tag-optimized encoder
|
||||
},
|
||||
{
|
||||
field: "alias", // Index the alias field for multi-language support
|
||||
tokenize: "full", // Full tokenization for complete alias matching
|
||||
encode: encodeAlias, // Use custom multi-language encoding function
|
||||
}
|
||||
]
|
||||
}
|
||||
});
|
||||
|
||||
return document;
|
||||
}
|
||||
|
||||
// Export functions for testing when in test environment
|
||||
const isTestEnvironment = typeof process !== 'undefined' && process.env.NODE_ENV === 'test';
|
||||
export const __test__ = isTestEnvironment ? { createTagEncoder, createCJKEncoder } : undefined;
|
||||
Reference in New Issue
Block a user