diff --git a/README.md b/README.md index f60a6ac..5cd6759 100644 --- a/README.md +++ b/README.md @@ -94,7 +94,7 @@ The main search screen remains usable as long as at least one indexed dictionary ## Tech Stack -- Kotlin 2.4.0 +- Kotlin 2.4.10 - Jetpack Compose (BOM 2026.06.01) - Material 3 - Coroutines and Flow 1.11.0 @@ -102,8 +102,8 @@ The main search screen remains usable as long as at least one indexed dictionary - Paging 3.5.0 - OkHttp 5.4.0 - kotlinx.serialization 1.11.0 -- AboutLibraries 15.0.3 metadata generation -- Android Gradle Plugin 9.2.1 +- AboutLibraries 15.0.4 metadata generation +- Android Gradle Plugin 9.3.0 ## Build diff --git a/app/build.gradle.kts b/app/build.gradle.kts index 8c72c16..14db2db 100644 --- a/app/build.gradle.kts +++ b/app/build.gradle.kts @@ -92,14 +92,8 @@ android { applicationId = "com.example.research" minSdk = project.property("minSdk").toString().toInt() targetSdk = project.property("targetSdk").toString().toInt() - versionCode = 3 - versionName = "1.1.1" - - val isFdroid = project.hasProperty("fdroid") - if (isFdroid) { - applicationIdSuffix = ".fdroid" - versionNameSuffix = "-fdroid" - } + versionCode = 4 + versionName = "1.2.0" vectorDrawables { useSupportLibrary = true diff --git a/app/src/main/java/com/example/research/ReSearchApplication.kt b/app/src/main/java/com/example/research/ReSearchApplication.kt index d2c5bb1..ce33077 100644 --- a/app/src/main/java/com/example/research/ReSearchApplication.kt +++ b/app/src/main/java/com/example/research/ReSearchApplication.kt @@ -45,7 +45,7 @@ class ReSearchApplication : Application() { applicationScope.launch(Dispatchers.IO) { preferencesManager.sanitizeDictionarySources() } - localDictionaryRepository = LocalDictionaryRepository(this) + localDictionaryRepository = LocalDictionaryRepository(this, preferencesManager = preferencesManager) okHttpClient = OkHttpClient.Builder() .protocols(listOf(Protocol.HTTP_1_1)) diff --git a/app/src/main/java/com/example/research/common/util/DateUtils.kt b/app/src/main/java/com/example/research/common/util/DateUtils.kt index d83fb67..3c7d203 100644 --- a/app/src/main/java/com/example/research/common/util/DateUtils.kt +++ b/app/src/main/java/com/example/research/common/util/DateUtils.kt @@ -13,7 +13,7 @@ object DateUtils { return dateFormat.format(calendar.time) } fun extractDateFromFileName(fileName: String): String? { - val regex = Regex("""_(\d{6})\.(gz|dsl|idx|dsl\.dz|dsl\.idx|dsl\.gz)$""") + val regex = Regex("""_(\d{6})\.(gz|dsl|idx|idx\.fold|dsl\.dz|dsl\.idx|dsl\.idx\.fold|dsl\.gz)$""") return regex.find(fileName)?.groupValues?.get(1) } fun isDateBefore(date1: String, date2: String): Boolean { diff --git a/app/src/main/java/com/example/research/core/domain/model/IndexEntry.kt b/app/src/main/java/com/example/research/core/domain/model/IndexEntry.kt index d79bb7d..04e99e8 100644 --- a/app/src/main/java/com/example/research/core/domain/model/IndexEntry.kt +++ b/app/src/main/java/com/example/research/core/domain/model/IndexEntry.kt @@ -13,7 +13,8 @@ data class IndexEntry( val offset: ArticleOffset = ArticleOffset.ZERO, val length: ArticleLength = ArticleLength.ZERO, val dictionaryName: String = "", - val dictionaryPath: String = "" + val dictionaryPath: String = "", + val isAlias: Boolean = false ) : Comparable { init { require(word.isNotBlank()) { "Word cannot be blank" } diff --git a/app/src/main/java/com/example/research/core/util/StringExtensions.kt b/app/src/main/java/com/example/research/core/util/StringExtensions.kt index 7f7f965..6bfc8ab 100644 --- a/app/src/main/java/com/example/research/core/util/StringExtensions.kt +++ b/app/src/main/java/com/example/research/core/util/StringExtensions.kt @@ -9,6 +9,32 @@ fun String.sanitizeQuery(): String { .take(MAX_QUERY_LENGTH) } +fun String.foldLatinDiacritics(): String { + val decomposed = java.text.Normalizer.normalize(this, java.text.Normalizer.Form.NFD) + val folded = StringBuilder(decomposed.length) + var followsLatinBase = false + var index = 0 + + while (index < decomposed.length) { + val codePoint = decomposed.codePointAt(index) + val type = Character.getType(codePoint) + val isMark = type == Character.NON_SPACING_MARK.toInt() || + type == Character.COMBINING_SPACING_MARK.toInt() || + type == Character.ENCLOSING_MARK.toInt() + + if (!(isMark && followsLatinBase)) { + folded.appendCodePoint(codePoint) + } + + if (!isMark) { + followsLatinBase = Character.UnicodeScript.of(codePoint) == Character.UnicodeScript.LATIN + } + index += Character.charCount(codePoint) + } + + return java.text.Normalizer.normalize(folded, java.text.Normalizer.Form.NFC) +} + fun String.sanitizeCacheKey(): String { return this.replace(":", "_").replace("\n", "_").replace("\t", "_") } diff --git a/app/src/main/java/com/example/research/data/local/preferences/PreferencesManager.kt b/app/src/main/java/com/example/research/data/local/preferences/PreferencesManager.kt index a298a16..7896611 100644 --- a/app/src/main/java/com/example/research/data/local/preferences/PreferencesManager.kt +++ b/app/src/main/java/com/example/research/data/local/preferences/PreferencesManager.kt @@ -6,6 +6,7 @@ import androidx.datastore.preferences.core.Preferences import androidx.datastore.preferences.core.edit import androidx.datastore.preferences.core.booleanPreferencesKey import androidx.datastore.preferences.core.stringPreferencesKey +import androidx.datastore.preferences.core.stringSetPreferencesKey import androidx.datastore.preferences.preferencesDataStore import com.example.research.core.domain.model.DictionarySource import kotlinx.coroutines.flow.Flow @@ -23,6 +24,7 @@ class PreferencesManager(private val context: Context) { val IS_THEME_EXPANDED = booleanPreferencesKey("is_theme_expanded") val IS_LANGUAGE_EXPANDED = booleanPreferencesKey("is_language_expanded") val IS_DICTIONARIES_EXPANDED = booleanPreferencesKey("is_dictionaries_expanded") + val DISABLED_DICTIONARY_PATHS = stringSetPreferencesKey("disabled_dictionary_paths") } private val json = Json { ignoreUnknownKeys = true } @@ -60,6 +62,33 @@ class PreferencesManager(private val context: Context) { } } + val disabledDictionaryPaths: Flow> = context.dataStore.data + .map { preferences -> + preferences[PreferencesKeys.DISABLED_DICTIONARY_PATHS] ?: emptySet() + } + + suspend fun setDictionaryActive(path: String, isActive: Boolean) { + context.dataStore.edit { preferences -> + val disabledPaths = preferences[PreferencesKeys.DISABLED_DICTIONARY_PATHS] + .orEmpty() + .toMutableSet() + if (isActive) { + disabledPaths.remove(path) + } else { + disabledPaths.add(path) + } + if (disabledPaths.isEmpty()) { + preferences.remove(PreferencesKeys.DISABLED_DICTIONARY_PATHS) + } else { + preferences[PreferencesKeys.DISABLED_DICTIONARY_PATHS] = disabledPaths + } + } + } + + suspend fun removeDictionaryActiveState(path: String) { + setDictionaryActive(path, true) + } + suspend fun saveThemeMode(theme: String) { context.dataStore.edit { preferences -> preferences[PreferencesKeys.THEME_MODE] = theme diff --git a/app/src/main/java/com/example/research/data/paging/IndexEntryPagingSource.kt b/app/src/main/java/com/example/research/data/paging/IndexEntryPagingSource.kt index 0a58267..9584e0a 100644 --- a/app/src/main/java/com/example/research/data/paging/IndexEntryPagingSource.kt +++ b/app/src/main/java/com/example/research/data/paging/IndexEntryPagingSource.kt @@ -15,85 +15,180 @@ import kotlinx.coroutines.coroutineScope import kotlinx.coroutines.CancellationException import java.io.File +/** + * Two-phase search paging. + * + * Page one (key = null) is the ranked head: the capped, relevance-ordered result the + * search screen has always shown, merged across active dictionaries. Every range scan + * that hit its cap hands back a [IndexSearcher.RangeCursor]; those become the tail key. + * + * Tail pages continue the unfinished range scans directly from the index files and + * merge them alphabetically (k-way, by index word), so scrolling eventually surfaces + * every prefix match in every active dictionary. Articles already shown -- by the head + * or by an earlier tail page, including diacritic alias duplicates -- are filtered + * through [emittedArticles], which grows only as far as the user actually scrolls. + * + * A new query builds a new PagingSource (see SearchViewModel's flatMapLatest), so an + * in-flight head or tail load is cancelled by Paging when the user keeps typing. + */ class IndexEntryPagingSource( private val indexSearcher: IndexSearcher, private val dictionaries: List, private val query: String -) : PagingSource() { +) : PagingSource() { - private var searchResults: List? = null + data class StreamCursor( + val dictPosition: Int, + val cursor: IndexSearcher.RangeCursor + ) - override val jumpingSupported: Boolean = true + data class TailKey(val streams: List) - override suspend fun load(params: LoadParams): LoadResult { - val offset = params.key ?: 0 + private data class EmittedKey( + val dictPosition: Int, + val originalWord: String, + val offset: Long, + val length: Int + ) + private val emittedArticles = HashSet() + + private val activeDictionaries = dictionaries.withIndex().filter { it.value.isActive } + + override suspend fun load(params: LoadParams): LoadResult { return try { - val allResults = searchResults ?: ReSearchTrace.asyncSection(ReSearchTrace.SEARCH_PAGING) { - coroutineScope { - val rawResults = dictionaries - .filter { it.isActive } - .map { dict -> - async { - val indexFile = File(dict.indexPath) - if (!indexFile.exists()) return@async emptyList() - - try { - val results = indexSearcher.search( - pathOrUri = indexFile.absolutePath, - query = query, - includeSubstringMatches = false - ) - results.map { - it.withDictionary( - dict.metadata.name.ifBlank { dict.name }, - dict.path - ) - } - } catch (e: CancellationException) { - throw e - } catch (_: Exception) { - emptyList() - } - } - } - .awaitAll() - .flatten() - - // Precompute sorting keys to optimize comparator performance. - // entry.word is already lowercase from indexing. - val normalizedQuery = query.sanitizeQuery().lowercase().trim() - - rawResults.rankBySearchRelevance( - SearchRankingContext(normalizedQuery) - ) - } - }.also { searchResults = it } - - val totalSize = allResults.size - val startIndex = offset.coerceAtLeast(0).coerceAtMost(totalSize) - val endIndex = (startIndex + params.loadSize).coerceAtMost(totalSize) - val pageItems = allResults.subList(startIndex, endIndex) - val hasMore = endIndex < totalSize - val nextKey = if (hasMore) endIndex else null - val prevKey = if (offset == 0) null else (offset - params.loadSize).coerceAtLeast(0) - - LoadResult.Page( - data = pageItems, - prevKey = prevKey, - nextKey = nextKey - ) + val key = params.key + if (key == null) loadHead() else loadTail(key, params.loadSize) } catch (e: CancellationException) { throw e } catch (e: Exception) { LoadResult.Error(e) } } - - override fun getRefreshKey(state: PagingState): Int? { - return state.anchorPosition?.let { anchorPosition -> - state.closestPageToPosition(anchorPosition)?.prevKey?.plus(state.config.pageSize) - ?: state.closestPageToPosition(anchorPosition)?.nextKey?.minus(state.config.pageSize) + + private suspend fun loadHead(): LoadResult = + ReSearchTrace.asyncSection(ReSearchTrace.SEARCH_PAGING) { + coroutineScope { + val perDictionary = activeDictionaries + .map { (position, dict) -> + async { + val indexFile = File(dict.indexPath) + if (!indexFile.exists()) return@async null + + try { + val result = indexSearcher.searchWithCursors( + pathOrUri = indexFile.absolutePath, + query = query + ) + Triple(position, dict, result) + } catch (e: CancellationException) { + throw e + } catch (_: Exception) { + null + } + } + } + .awaitAll() + .filterNotNull() + + val rawResults = perDictionary.flatMap { (position, dict, result) -> + result.entries.map { entry -> + position to entry.withDictionary( + dict.metadata.name.ifBlank { dict.name }, + dict.path + ) + } + } + + val normalizedQuery = query.sanitizeQuery().lowercase().trim() + val ranked = rawResults.map { it.second } + .rankBySearchRelevance(SearchRankingContext(normalizedQuery)) + + rawResults.forEach { (position, entry) -> emittedArticles.add(entry.emittedKey(position)) } + + val streams = perDictionary.flatMap { (position, _, result) -> + result.cursors.map { StreamCursor(position, it) } + } + LoadResult.Page( + data = ranked, + prevKey = null, + nextKey = if (streams.isEmpty()) null else TailKey(streams) + ) + } + } + + private suspend fun loadTail(key: TailKey, loadSize: Int): LoadResult { + val streams = key.streams.map { TailStream(it.dictPosition, it.cursor) } + val page = ArrayList(loadSize) + val pageKeys = HashSet() + val chunkSize = maxOf(loadSize, MIN_CHUNK) + + while (page.size < loadSize) { + streams.forEach { it.ensureBuffered(chunkSize) } + val next = streams + .filter { it.hasBuffered } + .minByOrNull { it.peekWord } + ?: break + + val scanned = next.consume() + val dict = dictionaries[next.dictPosition] + val entry = scanned.entry.withDictionary( + dict.metadata.name.ifBlank { dict.name }, + dict.path + ) + val emittedKey = entry.emittedKey(next.dictPosition) + if (emittedKey !in emittedArticles && pageKeys.add(emittedKey)) { + page.add(entry) + } + } + + emittedArticles.addAll(pageKeys) + + val remaining = streams.mapNotNull { stream -> + stream.nextCursor?.let { StreamCursor(stream.dictPosition, it) } + } + return LoadResult.Page( + data = page, + prevKey = null, + nextKey = if (remaining.isEmpty()) null else TailKey(remaining) + ) + } + + private inner class TailStream( + val dictPosition: Int, + initialCursor: IndexSearcher.RangeCursor + ) { + var nextCursor: IndexSearcher.RangeCursor? = initialCursor + private set + private var buffer: List = emptyList() + private var position = 0 + + val hasBuffered: Boolean get() = position < buffer.size + val peekWord: String get() = buffer[position].entry.word + + suspend fun ensureBuffered(chunkSize: Int) { + if (hasBuffered) return + val cursor = nextCursor ?: return + val indexPath = File(dictionaries[dictPosition].indexPath).absolutePath + buffer = indexSearcher.scanTailChunk(indexPath, cursor, chunkSize) + position = 0 + if (buffer.isEmpty()) nextCursor = null + } + + fun consume(): IndexSearcher.ScannedEntry { + val scanned = buffer[position] + position++ + nextCursor = scanned.cursorAfter + return scanned } } + + private fun IndexEntry.emittedKey(dictPosition: Int) = + EmittedKey(dictPosition, originalWord, offset.value, length.value) + + override fun getRefreshKey(state: PagingState): TailKey? { + return null + } } + +private const val MIN_CHUNK = 48 diff --git a/app/src/main/java/com/example/research/data/parser/DslCharsetDetector.kt b/app/src/main/java/com/example/research/data/parser/DslCharsetDetector.kt index dfd2aa2..237163f 100644 --- a/app/src/main/java/com/example/research/data/parser/DslCharsetDetector.kt +++ b/app/src/main/java/com/example/research/data/parser/DslCharsetDetector.kt @@ -58,17 +58,23 @@ object DslCharsetDetector { private fun looksLikeUtf8(inputStream: java.io.InputStream): Boolean { if (!inputStream.markSupported()) return false inputStream.mark(8192) - return try { + try { val sample = ByteArray(8192) val read = inputStream.read(sample) if (read <= 0) return false val decoder = Charsets.UTF_8.newDecoder() .onMalformedInput(java.nio.charset.CodingErrorAction.REPORT) .onUnmappableCharacter(java.nio.charset.CodingErrorAction.REPORT) - decoder.decode(java.nio.ByteBuffer.wrap(sample, 0, read)) - true + val byteBuffer = java.nio.ByteBuffer.wrap(sample, 0, read) + val charBuffer = java.nio.CharBuffer.allocate(8192) + while (true) { + val result = decoder.decode(byteBuffer, charBuffer, false) + if (result.isError) return false + if (result.isUnderflow) return true + charBuffer.clear() + } } catch (_: Exception) { - false + return false } finally { inputStream.reset() } diff --git a/app/src/main/java/com/example/research/data/parser/DslIndexer.kt b/app/src/main/java/com/example/research/data/parser/DslIndexer.kt index 9c8c437..5d4b4f5 100644 --- a/app/src/main/java/com/example/research/data/parser/DslIndexer.kt +++ b/app/src/main/java/com/example/research/data/parser/DslIndexer.kt @@ -4,6 +4,7 @@ import android.util.Log import com.example.research.core.domain.model.ArticleLength import com.example.research.core.domain.model.ArticleOffset import com.example.research.core.domain.model.IndexEntry +import com.example.research.core.util.foldLatinDiacritics import com.example.research.core.util.removeAccentTagsForIndexing import kotlinx.coroutines.Deferred import kotlinx.coroutines.Dispatchers @@ -23,7 +24,7 @@ class DslIndexer(context: Context? = null) { companion object { private const val TAG = "DslIndexer" - const val INDEX_VERSION = 4 + const val INDEX_VERSION = 6 const val BUFFER_SIZE = 1048576 const val BATCH_BUFFER_SIZE = 524288 private const val INDEX_SORT_PARALLELISM = 1 @@ -40,6 +41,9 @@ class DslIndexer(context: Context? = null) { private const val IN_MEMORY_SORT_MAX_ENTRIES = 140_000 private const val ESTIMATED_IN_MEMORY_ENTRY_BYTES = 768L private const val IN_MEMORY_SORT_HEAP_RESERVE_BYTES = 96L * 1024L * 1024L + + fun foldedIndexFile(indexFile: File): File = + File(indexFile.parent ?: ".", "${indexFile.name}.fold") } suspend fun createIndex( @@ -56,21 +60,25 @@ class DslIndexer(context: Context? = null) { java.io.BufferedInputStream(dslInputStream) } val tempFile = File(indexFile.parent, "${indexFile.name}.tmp") - val entryCount: Int + val counts: ParsedCounts try { + try { foldedIndexFile(indexFile).delete() } catch (_: Throwable) { /* Legacy v5 leftover */ } onOperationChange("Parsing file") onFileProgress(0.0f) - entryCount = parseDslToTempFile(bufferedStream, tempFile, + counts = parseDslToTempFile(bufferedStream, tempFile, onProgress, onFileProgress, onCharsetDetected) onFileProgress(0.5f) onOperationChange("Sorting index") - sortAndWriteIndexFile(tempFile, indexFile, entryCount, onOperationChange, onFileProgress) + sortAndWriteIndexFile(tempFile, indexFile, counts, onOperationChange, onFileProgress) + onFileProgress(1f) } finally { try { tempFile.delete() } catch (_: Throwable) { /* Ignore cleanup failure */ } } - try { onProgress(entryCount) } catch (_: Throwable) { /* Ignore progress callback failure */ } - return entryCount + try { onProgress(counts.articleCount) } catch (_: Throwable) { /* Ignore progress callback failure */ } + return counts.articleCount } + + private data class ParsedCounts(val articleCount: Int, val totalEntries: Int) // Blocking file I/O is safe across this file: every suspend fun here is reached only // from createIndex(), which the sole caller (KotlinDictionaryEngine) runs inside // withContext(Dispatchers.IO). The dispatcher is invisible across the suspend-call @@ -83,7 +91,7 @@ class DslIndexer(context: Context? = null) { onProgress: (Int) -> Unit, onFileProgress: (Float) -> Unit = {}, onCharsetDetected: (DslCharsetDetector.DetectedCharset) -> Unit = {} - ): Int { + ): ParsedCounts { val detectedCharset = DslCharsetDetector.detect(inputStream) onCharsetDetected(detectedCharset) @@ -98,6 +106,7 @@ class DslIndexer(context: Context? = null) { ) var entryCount = 0 + var aliasCount = 0 var lineCount = 0 var headwordLineCount = 0 var bodyLineCount = 0 @@ -127,9 +136,10 @@ class DslIndexer(context: Context? = null) { fun normalizeSearchableWord(searchableText: String, fallback: String): String { // Parser already lowercases and strips accent tags for non-empty results; // only the raw-headword fallback still needs normalization. - return searchableText.ifEmpty { + val normalized = searchableText.ifEmpty { fallback.removeAccentTagsForIndexing().lowercase() } + return java.text.Normalizer.normalize(normalized, java.text.Normalizer.Form.NFC) } DataOutputStream(BufferedOutputStream(FileOutputStream(tempFile), BUFFER_SIZE)).use { output -> @@ -144,12 +154,23 @@ class DslIndexer(context: Context? = null) { return } val length = (groupEndOffset - groupStartOffset).coerceAtLeast(0L).toInt() - for (idx in groupSearchable.indices) { - writer.write(groupSearchable[idx]) - writer.write(groupDisplay[idx]) + fun writeTempEntry(word: String, display: String, isAlias: Boolean) { + writer.write(word) + writer.write(display) output.writeLong(groupStartOffset) output.writeInt(length) + output.writeByte(if (isAlias) ENTRY_FLAG_ALIAS else 0) + } + for (idx in groupSearchable.indices) { + val searchableWord = groupSearchable[idx] + val displayWord = groupDisplay[idx] + writeTempEntry(searchableWord, displayWord, isAlias = false) entryCount++ + val foldedWord = searchableWord.foldLatinDiacritics() + if (foldedWord != searchableWord) { + writeTempEntry(foldedWord, displayWord, isAlias = true) + aliasCount++ + } } groupSearchable.clear() groupDisplay.clear() @@ -230,26 +251,28 @@ class DslIndexer(context: Context? = null) { try { onProgress(entryCount) } catch (_: Throwable) { /* Ignore progress callback failure */ } } onFileProgress(0.5f) - return entryCount + return ParsedCounts(articleCount = entryCount, totalEntries = entryCount + aliasCount) } // Blocking file I/O safe: reached only from createIndex() (runs on Dispatchers.IO). See parseDslToTempFile(). @Suppress("BlockingMethodInNonBlockingContext") - private suspend fun sortAndWriteIndexFile(tempFile: File, indexFile: File, entryCount: Int, onOperationChange: (String) -> Unit = {}, onFileProgress: (Float) -> Unit = {}) { - if (entryCount == 0) { + private suspend fun sortAndWriteIndexFile(tempFile: File, indexFile: File, counts: ParsedCounts, onOperationChange: (String) -> Unit = {}, onFileProgress: (Float) -> Unit = {}) { + if (counts.totalEntries == 0) { DataOutputStream(BufferedOutputStream(FileOutputStream(indexFile))).use { output -> output.writeInt(INDEX_VERSION) output.writeInt(0) output.writeInt(0) + output.writeInt(0) } return } - if (shouldUseInMemorySort(entryCount)) { - inMemorySortAndWrite(tempFile, indexFile, entryCount) + if (shouldUseInMemorySort(counts.totalEntries)) { + inMemorySortAndWrite(tempFile, indexFile, counts) onFileProgress(FINAL_INDEX_DONE_PROGRESS) } else { - externalSortAndWrite(tempFile, indexFile, entryCount, onOperationChange, onFileProgress) + externalSortAndWrite(tempFile, indexFile, counts, onOperationChange, onFileProgress) } } + private fun shouldUseInMemorySort(entryCount: Int): Boolean { if (entryCount <= IN_MEMORY_SORT_ALWAYS_ENTRIES) return true if (entryCount > IN_MEMORY_SORT_MAX_ENTRIES) return false @@ -264,30 +287,22 @@ class DslIndexer(context: Context? = null) { } // Blocking file I/O safe: reached only from createIndex() (runs on Dispatchers.IO). See parseDslToTempFile(). @Suppress("BlockingMethodInNonBlockingContext") - private suspend fun inMemorySortAndWrite(tempFile: File, indexFile: File, entryCount: Int) { - val entries = ArrayList(entryCount) + private suspend fun inMemorySortAndWrite(tempFile: File, indexFile: File, counts: ParsedCounts) { + val entries = ArrayList(counts.totalEntries) java.io.DataInputStream(java.io.BufferedInputStream(java.io.FileInputStream(tempFile), BATCH_BUFFER_SIZE)).use { input -> - repeat(entryCount) { - val word = input.readUtf8String() - val originalWord = input.readUtf8String() - val offset = input.readLong() - val length = input.readInt() - entries.add(IndexEntry( - word = word, - originalWord = originalWord, - offset = ArticleOffset(offset), - length = ArticleLength(length) - )) + repeat(counts.totalEntries) { + entries.add(input.readTempEntry()) if (entries.size % 5000 == 0) { try { yield() } catch (_: Throwable) { /* Ignore yield failure */ } } } } entries.sort() - writeIndexFile(indexFile, entries) + writeIndexFile(indexFile, entries, counts.articleCount) entries.clear() } - private suspend fun externalSortAndWrite(tempFile: File, indexFile: File, entryCount: Int, onOperationChange: (String) -> Unit, onFileProgress: (Float) -> Unit) { + private suspend fun externalSortAndWrite(tempFile: File, indexFile: File, counts: ParsedCounts, onOperationChange: (String) -> Unit, onFileProgress: (Float) -> Unit) { + val entryCount = counts.totalEntries val adaptiveConfig = resourceManager?.calculateAdaptiveConfig(estimatedEntryCount = entryCount.toLong()) ?: ResourceManager.AdaptiveConfig( batchSize = 100_000, @@ -317,16 +332,7 @@ class DslIndexer(context: Context? = null) { val batch = ArrayList(currentBatchSize) val batchEnd = minOf(processed + currentBatchSize, entryCount) while (processed < batchEnd) { - val word = input.readUtf8String() - val originalWord = input.readUtf8String() - val offset = input.readLong() - val length = input.readInt() - batch.add(IndexEntry( - word = word, - originalWord = originalWord, - offset = ArticleOffset(offset), - length = ArticleLength(length) - )) + batch.add(input.readTempEntry()) processed++ } @@ -422,7 +428,7 @@ class DslIndexer(context: Context? = null) { onFileProgress(MERGE_START_PROGRESS) onOperationChange("Merging ${sortedBatches.size} batches") - mergeSortedBatches(sortedBatches, indexFile, entryCount, onOperationChange, onFileProgress) + mergeSortedBatches(sortedBatches, indexFile, counts, onOperationChange, onFileProgress) onFileProgress(FINAL_INDEX_DONE_PROGRESS) } finally { sortedBatches.forEach { try { it.delete() } catch (_: Throwable) { /* Ignore cleanup failure */ } } @@ -443,6 +449,7 @@ class DslIndexer(context: Context? = null) { writer.write(entry.originalWord) output.writeLong(entry.offset.value) output.writeInt(entry.length.value) + output.writeByte(if (entry.isAlias) ENTRY_FLAG_ALIAS else 0) } } } @@ -453,15 +460,16 @@ class DslIndexer(context: Context? = null) { private suspend fun mergeSortedBatches( batchFiles: List, indexFile: File, - totalEntries: Int, + counts: ParsedCounts, onOperationChange: (String) -> Unit = {}, onFileProgress: (Float) -> Unit, ) { if (batchFiles.isEmpty()) return + val totalEntries = counts.totalEntries val sparsePoints = mutableListOf() val prefixCounts = HashMap(minOf(totalEntries / 10, 100_000)) - val headerSize = 4 + 4 + 4 + val headerSize = 4 + 4 + 4 + 4 val dataScratchFile = File(indexFile.parent, "${indexFile.name}.data") val pq = PriorityQueue() @@ -508,7 +516,8 @@ class DslIndexer(context: Context? = null) { val origSize = writer.write(entry.originalWord) dataOutput.writeLong(entry.offset.value) dataOutput.writeInt(entry.length.value) - currentByteOffset += 2 + wordSize + 2 + origSize + 8 + 4 + dataOutput.writeByte(if (entry.isAlias) ENTRY_FLAG_ALIAS else 0) + currentByteOffset += 2 + wordSize + 2 + origSize + 8 + 4 + 1 entryIndex++ @@ -542,6 +551,7 @@ class DslIndexer(context: Context? = null) { DataOutputStream(BufferedOutputStream(FileOutputStream(indexFile), BUFFER_SIZE)).use { output -> output.writeInt(INDEX_VERSION) output.writeInt(totalEntries) + output.writeInt(counts.articleCount) output.writeInt(sparsePoints.size) sparsePoints.forEachIndexed { idx, point -> @@ -632,18 +642,14 @@ class DslIndexer(context: Context? = null) { } private fun readNextEntry(input: java.io.DataInputStream): IndexEntry? { return try { - val word = input.readUtf8String() - val originalWord = input.readUtf8String() - val offset = ArticleOffset(input.readLong()) - val length = ArticleLength(input.readInt()) - IndexEntry(word, originalWord, offset, length) + input.readTempEntry() } catch (_: java.io.EOFException) { null } } - private fun writeIndexFile(indexFile: File, entries: List) { + private fun writeIndexFile(indexFile: File, entries: List, articleCount: Int) { val sparseData = buildAdaptiveSparseIndex(entries) - val headerSize = 4 + 4 + 4 + val headerSize = 4 + 4 + 4 + 4 // Cache sparse table bytes to avoid redundant conversions val sparseTableBytes = sparseData.map { CachedBytes.from(it.word) } @@ -657,13 +663,14 @@ class DslIndexer(context: Context? = null) { // Calculate sizes without allocating byte arrays val wordSize = utf8ByteSize(entry.word) val origSize = utf8ByteSize(entry.originalWord) - currentByteOffset += 2 + wordSize + 2 + origSize + 8 + 4 + currentByteOffset += 2 + wordSize + 2 + origSize + 8 + 4 + 1 } DataOutputStream(BufferedOutputStream(FileOutputStream(indexFile))).use { output -> val writer = Utf8Writer(output) output.writeInt(INDEX_VERSION) output.writeInt(entries.size) + output.writeInt(articleCount) output.writeInt(sparseData.size) sparseData.forEach { point -> @@ -677,6 +684,7 @@ class DslIndexer(context: Context? = null) { writer.write(entry.originalWord) output.writeLong(entry.offset.value) output.writeInt(entry.length.value) + output.writeByte(if (entry.isAlias) ENTRY_FLAG_ALIAS else 0) } } } @@ -707,7 +715,7 @@ class DslIndexer(context: Context? = null) { // Calculate sizes without allocating byte arrays val wordSize = utf8ByteSize(entry.word) val origSize = utf8ByteSize(entry.originalWord) - currentByteOffset += 2 + wordSize + 2 + origSize + 8 + 4 + currentByteOffset += 2 + wordSize + 2 + origSize + 8 + 4 + 1 } return sparsePoints @@ -824,6 +832,24 @@ private fun java.io.DataInputStream.readUtf8String(): String { readFully(bytes) return String(bytes, Charsets.UTF_8) } + +internal const val ENTRY_FLAG_ALIAS = 1 + +private fun java.io.DataInputStream.readTempEntry(): IndexEntry { + val word = readUtf8String() + val originalWord = readUtf8String() + val offset = readLong() + val length = readInt() + val flags = readUnsignedByte() + return IndexEntry( + word = word, + originalWord = originalWord, + offset = ArticleOffset(offset), + length = ArticleLength(length), + isAlias = flags and ENTRY_FLAG_ALIAS != 0 + ) +} + private enum class LineKind { HEADWORD, BODY, diff --git a/app/src/main/java/com/example/research/data/parser/IndexSearcher.kt b/app/src/main/java/com/example/research/data/parser/IndexSearcher.kt index e667d8a..881602d 100644 --- a/app/src/main/java/com/example/research/data/parser/IndexSearcher.kt +++ b/app/src/main/java/com/example/research/data/parser/IndexSearcher.kt @@ -5,6 +5,7 @@ import com.example.research.R import com.example.research.core.domain.model.ArticleLength import com.example.research.core.domain.model.ArticleOffset import com.example.research.core.domain.model.IndexEntry +import com.example.research.core.util.foldLatinDiacritics import com.example.research.core.util.DictionaryError import com.example.research.core.util.sanitizeCacheKey import com.example.research.core.util.sanitizeQuery @@ -23,7 +24,8 @@ import java.io.RandomAccessFile class IndexSearcher(private val context: Context) { private val metadataCache = MetadataLruCache(maxBytes = 64L * 1024L * 1024L) - private val resultsCache = ResultsLruCache(maxEntries = 20) + private val resultsCache = ResultsLruCache>(maxEntries = 20) + private val cursorResultsCache = ResultsLruCache(maxEntries = 20) /** * Pre-allocated reusable buffer for readEntry() to reduce allocations. @@ -49,6 +51,27 @@ class IndexSearcher(private val context: Context) { val comparedCount: Int ) + data class RangeCursor( + val rangeQuery: String, + val absoluteIndex: Int, + val fileOffset: Long + ) + + internal class RangeScanResult( + val entries: List, + val nextCursor: RangeCursor? + ) + + class CursorSearchResult( + val entries: List, + val cursors: List + ) + + class ScannedEntry( + val entry: IndexEntry, + val cursorAfter: RangeCursor? + ) + class IndexComparisonException(message: String) : IllegalStateException(message) suspend fun findFirstEntry(pathOrUri: String, query: String): IndexEntry? = withContext(Dispatchers.Default) { @@ -66,7 +89,7 @@ class IndexSearcher(private val context: Context) { var iterationCount = 0 while (currentIdx < count) { if (iterationCount % 1000 == 0) yield() - val nextEntry = readEntry(source) + val nextEntry = readEntry(source, metadata) val nextWordLower = nextEntry.word.lowercase() if (!nextWordLower.startsWith(normalizedQuery)) break if (nextWordLower == normalizedQuery) return@withContext nextEntry.copy(word = nextWordLower) @@ -93,6 +116,7 @@ class IndexSearcher(private val context: Context) { return@withContext emptyList() } val normalizedQuery = query.sanitizeQuery().lowercase().trim() + val foldedQuery = normalizedQuery.foldLatinDiacritics() val cacheKey = "$pathOrUri:${normalizedQuery.sanitizeCacheKey()}" resultsCache.get(cacheKey)?.let { return@withContext it @@ -102,33 +126,34 @@ class IndexSearcher(private val context: Context) { val count = metadata.totalCount if (count == 0) return@withContext emptyList() val results = openSource(pathOrUri).use { source -> - val prefixMatches = mutableListOf() - val substringMatches = mutableListOf() - val firstMatch = findFirstMatch(source, metadata, normalizedQuery) - if (firstMatch != null) { - prefixMatches.addAll(readMatchingEntries(source, count, firstMatch, normalizedQuery)) + val prefixMatches = readPrefixMatches(source, metadata, normalizedQuery).entries.toMutableList() + if (foldedQuery != normalizedQuery) { + prefixMatches.addAll(readPrefixMatches(source, metadata, foldedQuery).entries) } + + val substringMatches = mutableListOf() + val seenArticles = prefixMatches.mapTo(HashSet()) { it.articleKey() } if (includeSubstringMatches && count <= SUBSTRING_SCAN_MAX_ENTRIES && - prefixMatches.size < 100 && + seenArticles.size < MAX_SUBSTRING_RESULTS && normalizedQuery.length >= 2 ) { - val prefixMatchWords = prefixMatches.map { it.word }.toSet() source.seek(metadata.dataStartOffset) for (i in 0 until count) { if (i % 1000 == 0) yield() - if (substringMatches.size >= 100) break - val entry = readEntry(source) + if (substringMatches.size >= MAX_SUBSTRING_RESULTS) break + val entry = readEntry(source, metadata) val entryWordLower = entry.word.lowercase() - if (entryWordLower.contains(normalizedQuery) && entryWordLower !in prefixMatchWords) { + if (entryWordLower.contains(normalizedQuery) && + seenArticles.add(entry.articleKey()) + ) { substringMatches.add(entry.copy(word = entryWordLower)) } } } - val combinedResults = prefixMatches + substringMatches - combinedResults.rankBySearchRelevance( - SearchRankingContext(normalizedQuery) - ) + (prefixMatches + substringMatches) + .rankBySearchRelevance(SearchRankingContext(normalizedQuery, foldedQuery)) + .distinctBy { it.articleKey() } } resultsCache.put(cacheKey, results) results @@ -148,6 +173,77 @@ class IndexSearcher(private val context: Context) { metadataCache.getOrPut(pathOrUri) { loadMetadata(pathOrUri) } } + suspend fun searchWithCursors( + pathOrUri: String, + query: String + ): CursorSearchResult = withContext(Dispatchers.Default) { + if (query.isBlank()) { + return@withContext CursorSearchResult(emptyList(), emptyList()) + } + val normalizedQuery = query.sanitizeQuery().lowercase().trim() + val foldedQuery = normalizedQuery.foldLatinDiacritics() + val cacheKey = "$pathOrUri:${normalizedQuery.sanitizeCacheKey()}" + cursorResultsCache.get(cacheKey)?.let { + return@withContext it + } + try { + val metadata = metadataCache.getOrPut(pathOrUri) { loadMetadata(pathOrUri) } + if (metadata.totalCount == 0) { + return@withContext CursorSearchResult(emptyList(), emptyList()) + } + val result = openSource(pathOrUri).use { source -> + val scans = buildList { + add(readPrefixMatches(source, metadata, normalizedQuery)) + if (foldedQuery != normalizedQuery) { + add(readPrefixMatches(source, metadata, foldedQuery)) + } + } + val entries = scans.flatMap { it.entries } + .rankBySearchRelevance(SearchRankingContext(normalizedQuery, foldedQuery)) + .distinctBy { it.articleKey() } + CursorSearchResult(entries, scans.mapNotNull { it.nextCursor }) + } + cursorResultsCache.put(cacheKey, result) + result + } catch (e: CancellationException) { + throw e + } catch (e: Exception) { + Log.e(TAG, "searchWithCursors() - error: ${e.message}", e) + if (e is DictionaryError) throw e + throw DictionaryError.SearchError.QueryFailed(context.getString(R.string.error_search_failed, query.take(20), pathOrUri.take(20)), e) + } + } + + suspend fun scanTailChunk( + pathOrUri: String, + cursor: RangeCursor, + maxEntries: Int + ): List = withContext(Dispatchers.Default) { + val metadata = metadataCache.getOrPut(pathOrUri) { loadMetadata(pathOrUri) } + if (cursor.absoluteIndex >= metadata.totalCount) return@withContext emptyList() + openSource(pathOrUri).use { source -> + source.seek(cursor.fileOffset) + val out = ArrayList(maxEntries) + var currentIdx = cursor.absoluteIndex + var iterationCount = 0 + while (currentIdx < metadata.totalCount && out.size < maxEntries) { + if (iterationCount % 1000 == 0) yield() + val entry = readEntry(source, metadata) + val entryWordLower = entry.word.lowercase() + if (!entryWordLower.startsWith(cursor.rangeQuery)) break + currentIdx++ + iterationCount++ + val cursorAfter = if (currentIdx < metadata.totalCount) { + RangeCursor(cursor.rangeQuery, currentIdx, source.position()) + } else { + null + } + out.add(ScannedEntry(entry.copy(word = entryWordLower), cursorAfter)) + } + out + } + } + suspend fun compareEntryGroups( expectedIndexPath: String, actualIndexPath: String, @@ -168,8 +264,8 @@ class IndexSearcher(private val context: Context) { expectedSource.seek(expectedMetadata.dataStartOffset) actualSource.seek(actualMetadata.dataStartOffset) - var expectedNext = readEntryOrNull(expectedSource, expectedMetadata.totalCount > 0) - var actualNext = readEntryOrNull(actualSource, actualMetadata.totalCount > 0) + var expectedNext = readEntryOrNull(expectedSource, expectedMetadata, expectedMetadata.totalCount > 0) + var actualNext = readEntryOrNull(actualSource, actualMetadata, actualMetadata.totalCount > 0) var comparedCount = 0 while (expectedNext != null && actualNext != null) { @@ -191,6 +287,7 @@ class IndexSearcher(private val context: Context) { comparedCount++ expectedNext = readEntryOrNull( expectedSource, + expectedMetadata, comparedCount < expectedMetadata.totalCount ) } @@ -203,6 +300,7 @@ class IndexSearcher(private val context: Context) { actualReadCount++ actualNext = readEntryOrNull( actualSource, + actualMetadata, actualReadCount < actualMetadata.totalCount ) } @@ -233,14 +331,18 @@ class IndexSearcher(private val context: Context) { } } - private fun readEntryOrNull(source: RandomAccessSource, shouldRead: Boolean): IndexEntry? = - if (shouldRead) readEntry(source) else null + private fun readEntryOrNull( + source: RandomAccessSource, + metadata: IndexMetadata, + shouldRead: Boolean + ): IndexEntry? = if (shouldRead) readEntry(source, metadata) else null fun trimMemory(level: Int) { metadataCache.trim(level) resultsCache.trim(level) + cursorResultsCache.trim(level) } - + suspend fun isMetadataLoaded(pathOrUri: String): Boolean { return metadataCache.contains(pathOrUri) } @@ -252,7 +354,13 @@ class IndexSearcher(private val context: Context) { val count = input.readInt() return@withContext if (version >= 2) { - loadMetadataV2(input, version, count) + val headerBase = if (version >= 6) { + input.readInt() + 12L + } else { + 8L + } + loadMetadataV2(input, version, count, headerBase) } else { loadMetadataV1(input, version, count) } @@ -272,13 +380,13 @@ class IndexSearcher(private val context: Context) { // runs inside withContext(Dispatchers.IO). The dispatcher is not visible across the // suspend-call boundary, so the inspection is suppressed deliberately. @Suppress("BlockingMethodInNonBlockingContext") - private suspend fun loadMetadataV2(input: DataInputStream, version: Int, count: Int): IndexMetadata { + private suspend fun loadMetadataV2(input: DataInputStream, version: Int, count: Int, headerBase: Long): IndexMetadata { val sparseCount = input.readInt() val sparseIndices = IntArray(sparseCount) val sparseOffsets = LongArray(sparseCount) val sparseWords = Array(sparseCount) { "" } - var headerSize = 4 + 4 + 4L + var headerSize = headerBase + 4L for (i in 0 until sparseCount) { if (i % 1000 == 0) yield() @@ -337,6 +445,17 @@ class IndexSearcher(private val context: Context) { dataStartOffset = 8L ) } + + private suspend fun readPrefixMatches( + source: RandomAccessSource, + metadata: IndexMetadata, + query: String + ): RangeScanResult { + val firstMatch = findFirstMatch(source, metadata, query) + ?: return RangeScanResult(emptyList(), null) + return readMatchingEntries(source, metadata, firstMatch, query) + } + private suspend fun findFirstMatch(source: RandomAccessSource, metadata: IndexMetadata, query: String): Pair? { val sparseIndices = metadata.sparseIndices val sparseOffsets = metadata.sparseOffsets @@ -359,7 +478,7 @@ class IndexSearcher(private val context: Context) { while (currentAbsoluteIdx < metadata.totalCount) { if (iterationCount % 1000 == 0) yield() - val entry = readEntry(source) + val entry = readEntry(source, metadata) val entryWordLower = entry.word.lowercase() if (entryWordLower >= query) { return if (entryWordLower.startsWith(query)) Pair(currentAbsoluteIdx, entry.copy(word = entryWordLower)) else null @@ -394,53 +513,64 @@ class IndexSearcher(private val context: Context) { } private suspend fun readMatchingEntries( source: RandomAccessSource, - totalCount: Int, + metadata: IndexMetadata, firstMatch: Pair, query: String - ): List { + ): RangeScanResult { val results = mutableListOf() results.add(firstMatch.second) + val uniqueArticles = hashSetOf(firstMatch.second.articleKey()) var currentIdx = firstMatch.first + 1 var iterationCount = 0 - while (currentIdx < totalCount && results.size < 100) { + while (currentIdx < metadata.totalCount) { + if (uniqueArticles.size >= MAX_PREFIX_RESULTS) { + return RangeScanResult(results, RangeCursor(query, currentIdx, source.position())) + } if (iterationCount % 1000 == 0) yield() - val entry = readEntry(source) + val entry = readEntry(source, metadata) val entryWordLower = entry.word.lowercase() if (!entryWordLower.startsWith(query)) break + uniqueArticles.add(entry.articleKey()) results.add(entry.copy(word = entryWordLower)) currentIdx++ iterationCount++ } - return results + return RangeScanResult(results, null) } - private fun readEntry(source: RandomAccessSource): IndexEntry { - val wordLen = source.readUnsignedShort() - val word = if (wordLen <= 4096) { + private fun readPooledString(source: RandomAccessSource, length: Int): String { + return if (length <= 4096) { val buffer = entryBuffer.get() ?: throw IllegalStateException("Entry buffer pool exhausted") - source.readFully(buffer, 0, wordLen) - String(buffer, 0, wordLen, Charsets.UTF_8) + source.readFully(buffer, 0, length) + String(buffer, 0, length, Charsets.UTF_8) } else { - val bytes = ByteArray(wordLen) - source.readFully(bytes) - String(bytes, Charsets.UTF_8) - } - - val origLen = source.readUnsignedShort() - val originalWord = if (origLen <= 4096) { - val buffer = entryBuffer.get() ?: throw IllegalStateException("Entry buffer pool exhausted") - source.readFully(buffer, 0, origLen) - String(buffer, 0, origLen, Charsets.UTF_8) - } else { - val bytes = ByteArray(origLen) + val bytes = ByteArray(length) source.readFully(bytes) String(bytes, Charsets.UTF_8) } + } + private fun readEntry(source: RandomAccessSource, metadata: IndexMetadata): IndexEntry { + val word = readPooledString(source, source.readUnsignedShort()) + val originalWord = readPooledString(source, source.readUnsignedShort()) val offset = ArticleOffset(source.readLong()) val length = ArticleLength(source.readInt()) - return IndexEntry(word, originalWord, offset, length) + val isAlias = if (metadata.version >= 6) { + source.readUnsignedByte() and ENTRY_FLAG_ALIAS != 0 + } else { + false + } + return IndexEntry(word, originalWord, offset, length, isAlias = isAlias) } + + private data class ArticleKey( + val originalWord: String, + val offset: Long, + val length: Int + ) + + private fun IndexEntry.articleKey() = ArticleKey(originalWord, offset.value, length.value) + private fun openSource(pathOrUri: String): RandomAccessSource { val file = File(pathOrUri) if (!file.exists()) throw DictionaryError.FileSystemError.NotFound(pathOrUri) @@ -455,20 +585,22 @@ class IndexSearcher(private val context: Context) { private const val TAG = "IndexSearcher" private const val SUBSTRING_SCAN_MAX_ENTRIES = 200_000 private const val SPARSE_INTERVAL_V1 = 128 + private const val MAX_PREFIX_RESULTS = 200 + private const val MAX_SUBSTRING_RESULTS = 100 } } -private class ResultsLruCache(private val maxEntries: Int) { +private class ResultsLruCache(private val maxEntries: Int) { private val mutex = Mutex() - private val map = object : LinkedHashMap>(maxEntries, 0.75f, true) { - override fun removeEldestEntry(eldest: MutableMap.MutableEntry>): Boolean { + private val map = object : LinkedHashMap(maxEntries, 0.75f, true) { + override fun removeEldestEntry(eldest: MutableMap.MutableEntry): Boolean { return size > maxEntries } } - suspend fun get(key: String): List? = mutex.withLock { + suspend fun get(key: String): V? = mutex.withLock { map[key] } - suspend fun put(key: String, value: List) = mutex.withLock { + suspend fun put(key: String, value: V) = mutex.withLock { map[key] = value } fun trim(level: Int) { @@ -548,6 +680,8 @@ private class MetadataLruCache(private val maxBytes: Long) { } interface RandomAccessSource : Closeable { fun seek(pos: Long) + fun position(): Long + fun readUnsignedByte(): Int fun readUnsignedShort(): Int fun readInt(): Int fun readLong(): Long @@ -561,6 +695,10 @@ class FileSource(file: File) : RandomAccessSource { raf.seek(pos) } + override fun position(): Long = raf.filePointer + + override fun readUnsignedByte(): Int = raf.readUnsignedByte() + override fun readUnsignedShort(): Int = raf.readUnsignedShort() override fun readInt(): Int = raf.readInt() diff --git a/app/src/main/java/com/example/research/data/parser/SearchRanking.kt b/app/src/main/java/com/example/research/data/parser/SearchRanking.kt index d1ca7da..4185480 100644 --- a/app/src/main/java/com/example/research/data/parser/SearchRanking.kt +++ b/app/src/main/java/com/example/research/data/parser/SearchRanking.kt @@ -1,15 +1,17 @@ package com.example.research.data.parser import com.example.research.core.domain.model.IndexEntry +import com.example.research.core.util.foldLatinDiacritics internal data class SearchRankingContext( val normalizedQuery: String, + val foldedQuery: String = normalizedQuery.foldLatinDiacritics(), ) private data class SortKey( - val charCount: Int, + val matchTier: Int, val exactMatch: Int, - val startsWithMatch: Int, + val charCount: Int, val word: String, val entry: IndexEntry, ) @@ -26,13 +28,18 @@ internal fun List.rankBySearchRelevance( .ifEmpty { entry.word.trim() } SortKey( - charCount = displayKey.codePointCount(0, displayKey.length), + matchTier = when { + !entry.isAlias && word.startsWith(ranking.normalizedQuery) -> 0 + entry.isAlias && word.startsWith(ranking.foldedQuery) -> 1 + !entry.isAlias && word.foldLatinDiacritics().startsWith(ranking.foldedQuery) -> 1 + else -> 2 + }, exactMatch = if (word == ranking.normalizedQuery) 0 else 1, - startsWithMatch = if (word.startsWith(ranking.normalizedQuery)) 0 else 1, + charCount = displayKey.codePointCount(0, displayKey.length), word = word, entry = entry, ) } - .sortedWith(compareBy({ it.charCount }, { it.exactMatch }, { it.startsWithMatch }, { it.word })) + .sortedWith(compareBy({ it.matchTier }, { it.exactMatch }, { it.charCount }, { it.word })) .map { it.entry } } diff --git a/app/src/main/java/com/example/research/data/repository/LocalDictionaryRepository.kt b/app/src/main/java/com/example/research/data/repository/LocalDictionaryRepository.kt index c3c5e86..51c53c6 100644 --- a/app/src/main/java/com/example/research/data/repository/LocalDictionaryRepository.kt +++ b/app/src/main/java/com/example/research/data/repository/LocalDictionaryRepository.kt @@ -14,6 +14,7 @@ import com.example.research.core.util.extractDictionaryPrefix import com.example.research.core.util.sanitizeQuery import com.example.research.core.util.toDictionaryError import com.example.research.data.dictzip.DictZipRandomAccessFile +import com.example.research.data.local.preferences.PreferencesManager import com.example.research.data.parser.DslCharsetDetector import com.example.research.data.parser.DslHeaderParser import com.example.research.data.parser.DslIndexer @@ -26,6 +27,7 @@ import kotlinx.coroutines.awaitAll import kotlinx.coroutines.ensureActive import kotlinx.coroutines.flow.MutableStateFlow import kotlinx.coroutines.flow.StateFlow +import kotlinx.coroutines.flow.first import kotlinx.coroutines.flow.update import kotlinx.coroutines.launch import kotlinx.coroutines.supervisorScope @@ -85,15 +87,11 @@ private class ProgressTracker(files: List>(emptyList()) @@ -107,6 +105,7 @@ class LocalDictionaryRepository( private val indexingMutex = Mutex() private val indexingSemaphore = Semaphore(1) private val warmupMutex = Mutex() + private val dictionaryStateMutex = Mutex() @Volatile private var currentIndexingJob: Job? = null @@ -140,7 +139,7 @@ class LocalDictionaryRepository( val filesToScan = scanLocalDirectory(pathOrUri) if (filesToScan.isEmpty()) { - mutableDictionaries.value = emptyList() + replaceDictionaries(emptyList()) mutableIndexingProgress.value = IndexingProgress(isIndexing = false) return@withContext OperationResult.Success(0) } @@ -159,12 +158,12 @@ class LocalDictionaryRepository( } if (filesToIndex.isEmpty()) { - mutableDictionaries.value = existingDictionaries.sortedBy { it.name } + replaceDictionaries(existingDictionaries) engine.getIndexSearcher().trimMemory(80) return@withContext OperationResult.Success(existingDictionaries.size) } - mutableDictionaries.value = existingDictionaries.sortedBy { it.name } + replaceDictionaries(existingDictionaries) val totalArticlesIndexed = AtomicInteger(existingDictionaries.sumOf { it.articleCount }) @@ -233,12 +232,9 @@ class LocalDictionaryRepository( if (res is OperationResult.Success) { progressTracker?.updateFileProgress(file.name, 1.0f) totalArticlesIndexed.addAndGet(res.data.articleCount) - mutableDictionaries.update { current -> - (current + res.data) - .distinctBy { it.path } - .sortedBy { it.name } - } - res.data + val dictionary = res.data + addDictionary(dictionary) + dictionary } else null } } @@ -266,17 +262,37 @@ class LocalDictionaryRepository( } } + private suspend fun replaceDictionaries(dictionaries: List) { + dictionaryStateMutex.withLock { applyDictionaries(dictionaries) } + } + + private suspend fun addDictionary(dictionary: Dictionary) { + dictionaryStateMutex.withLock { + applyDictionaries((mutableDictionaries.value + dictionary).distinctBy { it.path }) + } + } + + private suspend fun applyDictionaries(dictionaries: List) { + val disabledPaths = preferencesManager.disabledDictionaryPaths.first() + mutableDictionaries.value = dictionaries + .map { it.withActiveStatus(it.path !in disabledPaths) } + .sortedBy { it.name } + } + private suspend fun tryLoadExistingIndex(file: DiscoveredFile): Dictionary? { val dictFile = File(file.localPath) val indexFile = File(file.indexPath) - if (!indexFile.exists() || indexFile.length() == 0L) return null + if (!indexFile.exists() || indexFile.length() == 0L) { + deleteIndexFiles(indexFile) + return null + } if (!dictFile.exists()) return null val header = readIndexHeader(indexFile) if (header == null || header.version != DslIndexer.INDEX_VERSION) { - try { indexFile.delete() } catch (_: Exception) {} + deleteIndexFiles(indexFile) return null } @@ -326,16 +342,31 @@ class LocalDictionaryRepository( ) OperationResult.Success(dict) } else { - indexFile.delete() + deleteIndexFiles(indexFile) OperationResult.Error(result.errorMsg, null) } } catch(e: Exception) { - indexFile.delete() + deleteIndexFiles(indexFile) if (e is CancellationException) throw e OperationResult.Error(e.message ?: context.getString(R.string.error_indexing), e) } } + private fun deleteIndexFiles(indexFile: File, errors: MutableList? = null) { + deleteFileTracked(indexFile, "index file", errors) + deleteFileTracked(DslIndexer.foldedIndexFile(indexFile), "legacy folded index file", errors) + } + + private fun deleteFileTracked(file: File, label: String, errors: MutableList?) { + try { + if (file.exists() && !file.delete()) { + errors?.add("Failed to delete $label: ${file.absolutePath}") + } + } catch (e: Exception) { + errors?.add("Failed to delete $label: ${file.absolutePath} (${e.message})") + } + } + private fun scanLocalDirectory(path: String): List { val dir = File(path) if (!dir.exists() || !dir.isDirectory) return emptyList() @@ -354,7 +385,7 @@ class LocalDictionaryRepository( suspend fun search(query: String): OperationResult> = withContext(defaultDispatcher) { if (query.isBlank()) return@withContext OperationResult.Success(emptyList()) - val activeDicts = mutableDictionaries.value + val activeDicts = mutableDictionaries.value.filter { it.isActive } val sanitizedQuery = query.sanitizeQuery() ReSearchTrace.asyncSection(ReSearchTrace.SEARCH_DIRECT) { @@ -404,23 +435,33 @@ class LocalDictionaryRepository( return engine.getIndexSearcher() } - fun toggleDictionaryActive(dictionaryPath: String) { - mutableDictionaries.update { currentList -> - currentList.map { dict -> - if (dict.path == dictionaryPath) { - dict.withActiveStatus(!dict.isActive) - } else { - dict + suspend fun toggleDictionaryActive(dictionaryPath: String) { + dictionaryStateMutex.withLock { + val isActive = mutableDictionaries.value + .firstOrNull { it.path == dictionaryPath } + ?.isActive + ?.not() + ?: return + preferencesManager.setDictionaryActive(dictionaryPath, isActive) + mutableDictionaries.update { currentList -> + currentList.map { dict -> + if (dict.path == dictionaryPath) { + dict.withActiveStatus(isActive) + } else { + dict + } } } } } - fun deleteDictionary(dictionary: Dictionary): OperationResult { + private fun removeDictionaryFromState(path: String) { + mutableDictionaries.update { current -> current.filter { it.path != path } } + } + + suspend fun deleteDictionary(dictionary: Dictionary): OperationResult { try { - mutableDictionaries.update { current -> - current.filter { it.path != dictionary.path } - } + dictionaryStateMutex.withLock { removeDictionaryFromState(dictionary.path) } launch(ioDispatcher) { try { @@ -439,9 +480,12 @@ class LocalDictionaryRepository( File("${dictionary.indexPath}.idx") } - if (indexFile.exists()) { - if (!indexFile.delete()) { - deleteErrors.add("Failed to delete index file: ${indexFile.absolutePath}") + deleteIndexFiles(indexFile, deleteErrors) + + if (!dictFile.exists()) { + dictionaryStateMutex.withLock { + preferencesManager.removeDictionaryActiveState(dictionary.path) + removeDictionaryFromState(dictionary.path) } } @@ -507,13 +551,22 @@ class LocalDictionaryRepository( if (count < 0) return null + var articleCount = count + if (version >= 6) { + if (indexFile.length() < INDEX_V6_HEADER_BYTES) return null + articleCount = dataInput.readInt() + if (articleCount < 0) return null + } + if (version >= 2) { - if (indexFile.length() < INDEX_V2_HEADER_BYTES) return null + val fixedHeaderBytes = + if (version >= 6) INDEX_V6_HEADER_BYTES else INDEX_V2_HEADER_BYTES + if (indexFile.length() < fixedHeaderBytes) return null val sparseCount = dataInput.readInt() if (sparseCount < 0) return null - var headerBytes = INDEX_V2_HEADER_BYTES + var headerBytes = fixedHeaderBytes repeat(sparseCount) { dataInput.readInt() dataInput.readLong() @@ -524,7 +577,7 @@ class LocalDictionaryRepository( } } - IndexHeader(version, count) + IndexHeader(version, articleCount) } } } catch (e: Exception) { @@ -602,6 +655,7 @@ class LocalDictionaryRepository( const val MIN_PROGRESS_EMIT_INTERVAL_MS = 200L const val LEGACY_INDEX_HEADER_BYTES = 8L const val INDEX_V2_HEADER_BYTES = 12 + const val INDEX_V6_HEADER_BYTES = 16 const val SPARSE_ENTRY_FIXED_BYTES = 14 } } diff --git a/app/src/main/java/com/example/research/feature/download/repository/DictionaryCleaner.kt b/app/src/main/java/com/example/research/feature/download/repository/DictionaryCleaner.kt index 11f1776..2a1c6d3 100644 --- a/app/src/main/java/com/example/research/feature/download/repository/DictionaryCleaner.kt +++ b/app/src/main/java/com/example/research/feature/download/repository/DictionaryCleaner.kt @@ -59,11 +59,11 @@ class DictionaryCleaner( toDelete.add(item.name) continue } - if ((item.ext == "idx" || item.ext == "dsl.idx") && item.date != latestDsl) { + if (item.ext in indexExtensions && item.date != latestDsl) { toDelete.add(item.name) continue } - if ((item.ext == "idx" || item.ext == "dsl.idx") && item.date !in dslDates) { + if (item.ext in indexExtensions && item.date !in dslDates) { toDelete.add(item.name) } } @@ -76,10 +76,13 @@ class DictionaryCleaner( val date: String, val ext: String, ) - private val validExtensions = setOf("dsl", "dsl.dz", "dsl.gz", "gz", "idx", "dsl.idx") + private val indexExtensions = setOf("idx", "dsl.idx", "idx.fold", "dsl.idx.fold") + private val validExtensions = setOf("dsl", "dsl.dz", "dsl.gz", "gz") + indexExtensions private fun parseManagedName(name: String, prefixes: Set): ManagedFileName? { val ext = when { + name.endsWith(".dsl.idx.fold") -> "dsl.idx.fold" + name.endsWith(".idx.fold") -> "idx.fold" name.endsWith(".dsl.dz") -> "dsl.dz" name.endsWith(".dsl.gz") -> "dsl.gz" name.endsWith(".dsl.idx") -> "dsl.idx" diff --git a/app/src/main/java/com/example/research/feature/search/SearchViewModel.kt b/app/src/main/java/com/example/research/feature/search/SearchViewModel.kt index b607a7e..6079b04 100644 --- a/app/src/main/java/com/example/research/feature/search/SearchViewModel.kt +++ b/app/src/main/java/com/example/research/feature/search/SearchViewModel.kt @@ -98,8 +98,7 @@ class SearchViewModel( pageSize = 12, prefetchDistance = 1, enablePlaceholders = false, - initialLoadSize = 12, - jumpThreshold = 1 + initialLoadSize = 12 ), pagingSourceFactory = { IndexEntryPagingSource( diff --git a/app/src/main/java/com/example/research/ui/main/components/SearchResultsList.kt b/app/src/main/java/com/example/research/ui/main/components/SearchResultsList.kt index 7da8297..077fb3c 100644 --- a/app/src/main/java/com/example/research/ui/main/components/SearchResultsList.kt +++ b/app/src/main/java/com/example/research/ui/main/components/SearchResultsList.kt @@ -41,7 +41,6 @@ fun SearchResultsList( ) { items( count = results.itemCount, - key = results.itemKey { entry -> "${entry.dictionaryPath}_${entry.offset.value}_${entry.word}" }, contentType = results.itemContentType { "search_result" } ) { index -> results[index]?.let { entry -> diff --git a/app/src/main/java/com/example/research/ui/settings/SettingsViewModel.kt b/app/src/main/java/com/example/research/ui/settings/SettingsViewModel.kt index 20c00df..63bd181 100644 --- a/app/src/main/java/com/example/research/ui/settings/SettingsViewModel.kt +++ b/app/src/main/java/com/example/research/ui/settings/SettingsViewModel.kt @@ -331,7 +331,9 @@ class SettingsViewModel( } private fun toggleDictionaryActive(dictionaryPath: String) { - localDictionaryRepository.toggleDictionaryActive(dictionaryPath) + viewModelScope.launch { + localDictionaryRepository.toggleDictionaryActive(dictionaryPath) + } } private fun deleteDictionary(dictionary: Dictionary) { diff --git a/fastlane/metadata/android/en-US/changelogs/4.txt b/fastlane/metadata/android/en-US/changelogs/4.txt new file mode 100644 index 0000000..3e71108 --- /dev/null +++ b/fastlane/metadata/android/en-US/changelogs/4.txt @@ -0,0 +1,4 @@ +- Fixed a crash and lost dictionary state around search results +- Added diacritic-insensitive search, so accented and plain spellings match each other +- Fixed indexing failures on dictionaries with dense non-Latin text +- Search results now keep loading as you scroll instead of stopping early diff --git a/fastlane/metadata/android/ru-RU/changelogs/4.txt b/fastlane/metadata/android/ru-RU/changelogs/4.txt new file mode 100644 index 0000000..8e298ca --- /dev/null +++ b/fastlane/metadata/android/ru-RU/changelogs/4.txt @@ -0,0 +1,4 @@ +- Исправлен сбой и потеря состояния словаря в результатах поиска +- Добавлен поиск без учёта диакритических знаков — совпадают варианты с акцентами и без +- Исправлена ошибка индексации словарей с плотным нелатинским текстом +- Результаты поиска теперь полностью подгружаются при прокрутке, а не обрываются на первых совпадениях diff --git a/gradle/libs.versions.toml b/gradle/libs.versions.toml index b35e664..ef9b0e0 100644 --- a/gradle/libs.versions.toml +++ b/gradle/libs.versions.toml @@ -1,12 +1,12 @@ [versions] -aboutlibraries = "15.0.3" +aboutlibraries = "15.0.4" activity_compose = "1.13.0" -agp = "9.2.1" +agp = "9.3.0" compose_bom = "2026.06.01" core = "1.19.0" datastore_preferences = "1.2.1" documentfile = "1.1.0" -kotlin = "2.4.0" +kotlin = "2.4.10" lifecycle_runtime_ktx = "2.11.0" okhttp = "5.4.0" paging = "3.5.0" diff --git a/gradle/wrapper/gradle-wrapper.properties b/gradle/wrapper/gradle-wrapper.properties index eb84db6..a9db115 100644 --- a/gradle/wrapper/gradle-wrapper.properties +++ b/gradle/wrapper/gradle-wrapper.properties @@ -1,6 +1,6 @@ distributionBase=GRADLE_USER_HOME distributionPath=wrapper/dists -distributionUrl=https\://services.gradle.org/distributions/gradle-9.6.0-bin.zip +distributionUrl=https\://services.gradle.org/distributions/gradle-9.6.1-bin.zip networkTimeout=10000 retries=0 retryBackOffMs=500