diff --git a/.github/workflows/_build-macos.yml b/.github/workflows/_build-macos.yml index bff09ca89..987b7c95a 100644 --- a/.github/workflows/_build-macos.yml +++ b/.github/workflows/_build-macos.yml @@ -24,7 +24,7 @@ on: jobs: build: name: "Build for macOS" - runs-on: macos-latest + runs-on: macos-15 steps: - name: Checkout repository diff --git a/CHANGELOG.md b/CHANGELOG.md index 378462819..40a40dd59 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,11 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [1.41.1] - 2026-09-30 +### Fixed +- Keyword search now finds the keyword anywhere in the text, including inside words (`copyright` in `SPDX-FileCopyrightText`). Existing projects are reindexed when opened if their source code is available. +- Keyword search now indexes `vendor/`, `node_modules/` and dot-folders when **Include all file types** is enabled. + ## [1.41.0] - 2026-07-06 ### Added - **Import dependency identifications from another project.** The "Import identifications from…" flow now has an **Include dependencies** option that previews the source project's declared dependency identifications in a dedicated table and imports them alongside file/component identifications. Dependencies are matched by manifest path and PURL, and honor the same **Override previous work** toggle. @@ -281,3 +286,4 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 [1.40.0]: https://github.com/scanoss/sbom-workbench/compare/v1.39.2...v1.40.0 [1.40.1]: https://github.com/scanoss/sbom-workbench/compare/v1.40.0...v1.40.1 [1.41.0]: https://github.com/scanoss/sbom-workbench/compare/v1.40.1...v1.41.0 +[1.41.1]: https://github.com/scanoss/sbom-workbench/compare/v1.41.0...v1.41.1 diff --git a/release/app/package.json b/release/app/package.json index 6b126d487..535a71db0 100644 --- a/release/app/package.json +++ b/release/app/package.json @@ -1,6 +1,6 @@ { "name": "scanoss-workbench", - "version": "1.41.0", + "version": "1.41.1", "description": "Desktop version to use SCANOSS OSS in your projects", "license": "GPL-2.0-only", "author": { diff --git a/src/__tests__/search/indexer-searcher.test.ts b/src/__tests__/search/indexer-searcher.test.ts new file mode 100644 index 000000000..ced524d02 --- /dev/null +++ b/src/__tests__/search/indexer-searcher.test.ts @@ -0,0 +1,211 @@ +/** + * @jest-environment node + */ +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { Indexer } from '../../main/modules/searchEngine/indexer/Indexer'; +import { readIndexVersion, searcher } from '../../main/modules/searchEngine/searcher/Searcher'; +import { BlackListKeyWordIndex } from '../../main/workspace/tree/blackList/BlackListKeyWordIndex'; +import { + containsAllTerms, + getLegacySearchConfig, + getQueryTerms, + isExactIndexQuery, + SEARCH_INDEX_VERSION, +} from '../../shared/utils/search-utils'; + +jest.mock('electron-log', () => ({ info: jest.fn(), warn: jest.fn(), error: jest.fn() })); +jest.mock('../../main/broadcastManager/BroadcastManager', () => ({ + broadcastManager: { get: () => ({ send: jest.fn() }) }, +})); + +const { Index } = require('flexsearch'); + +const utf16 = (text: string) => Buffer.concat([Buffer.from([0xff, 0xfe]), Buffer.from(text, 'utf16le')]); + +const FIXTURES: Record = { + '/LICENSE': 'MIT License\n\nCopyright (c) 2020 Foo\n\nPermission is hereby granted', + '/src/main.c': 'int main() { return 0; } // Copyright (c) Bar', + '/src/header.ts': '// SPDX-FileCopyrightText: 2024 Acme\nexport const x = 1;', + '/src/crypto.js': 'const cipher = createEncryption(key);', + '/src/hash.go': 'sum := sha256.Sum256(data)', + '/docs/utf16.txt': utf16('hello utf16 world'), + '/img/logo.png': Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x00, 0x00, 0x1a, 0x0a]), + "/src/it's.txt": 'quoted path content', + '/docs/intl.md': 'Diseño del año, straße und 日本語テキスト', +}; + +describe('keyword search index', () => { + let root: string; + let source: string; + let dictionary: string; + const ids: Record = {}; + + const search = (query: string, params: Record = { limit: 1000 }) => ( + searcher.search({ query, params }).sort((a, b) => a - b) + ); + + beforeAll(async () => { + root = fs.mkdtempSync(path.join(os.tmpdir(), 'search-index-')); + source = path.join(root, 'source'); + dictionary = path.join(root, 'dictionary'); + const files = Object.entries(FIXTURES).map(([file, content], i) => { + fs.mkdirSync(path.join(source, path.dirname(file)), { recursive: true }); + fs.writeFileSync(path.join(source, file), content); + ids[file] = i + 1; + return { fileId: i + 1, path: file }; + }); + for (let i = 0; i < 600; i += 1) { + const file = `/bulk/file${i}.txt`; + fs.mkdirSync(path.join(source, 'bulk'), { recursive: true }); + fs.writeFileSync(path.join(source, file), `bulk content number ${i}`); + files.push({ fileId: 1000 + i, path: file }); + } + const indexer = new Indexer(); + await indexer.saveIndex(await indexer.index(files, source), dictionary); + searcher.closeIndex(); + searcher.loadIndex(dictionary); + }); + + afterAll(() => { + searcher.closeIndex(); + fs.rmSync(root, { recursive: true, force: true }); + }); + + it('writes the index version', () => { + expect(readIndexVersion(dictionary)).toBe(SEARCH_INDEX_VERSION); + expect(searcher.getVersion()).toBe(SEARCH_INDEX_VERSION); + }); + + it('finds keywords in extension-less files and inline comments', () => { + expect(search('copyright')).toEqual([ids['/LICENSE'], ids['/src/main.c'], ids['/src/header.ts']]); + }); + + it('finds 1 and 2 char keywords as whole words', () => { + expect(search('c')).toEqual([ids['/LICENSE'], ids['/src/main.c']]); + expect(search('x')).toEqual([ids['/src/header.ts']]); + }); + + it('finds keywords inside words', () => { + expect(search('crypt')).toEqual([ids['/src/crypto.js']]); + expect(search('sha')).toEqual([ids['/src/hash.go']]); + }); + + it('indexes UTF-16 files and skips binaries', () => { + expect(search('utf16')).toEqual([ids['/docs/utf16.txt']]); + expect(search('png')).toEqual([]); + }); + + it('indexes paths containing quotes', () => { + expect(search('quoted')).toEqual([ids["/src/it's.txt"]]); + }); + + it('finds non-ASCII keywords', () => { + expect(search(getQueryTerms('año').join(' '))).toEqual([ids['/docs/intl.md']]); + expect(search(getQueryTerms('日本語').join(' '))).toEqual([ids['/docs/intl.md']]); + }); + + it('removes temp folders left by a crashed save', async () => { + const other = path.join(root, 'other'); + fs.mkdirSync(other); + const orphan = path.join(other, 'dictionary.1-1.tmp'); + fs.mkdirSync(orphan); + const indexer = new Indexer(); + await indexer.saveIndex(await indexer.index([{ fileId: 1, path: '/LICENSE' }], source), path.join(other, 'dictionary')); + expect(fs.readdirSync(other)).toEqual(['dictionary']); + }); + + it('requires every term of a multi-term query', () => { + expect(search('copyright permission')).toEqual([ids['/LICENSE']]); + }); + + it('pages results with offset', () => { + const first = search('bulk', { limit: 500, offset: 0 }); + const second = search('bulk', { limit: 500, offset: 500 }); + expect(first).toHaveLength(500); + expect(second).toHaveLength(100); + expect(first.filter((id) => second.includes(id))).toEqual([]); + }); + + it('loads a legacy dictionary without a version file', () => { + const legacy = path.join(root, 'legacy'); + fs.mkdirSync(legacy); + const index = new Index(getLegacySearchConfig()); + index.add(1, 'legacy copyright notice'); + index.export((key: string, data: string) => fs.writeFileSync(path.join(legacy, `${key}.json`), data ?? '')); + searcher.closeIndex(); + searcher.loadIndex(legacy); + expect(searcher.getVersion()).toBe(1); + expect(searcher.search({ query: 'copyright' })).toEqual([1]); + expect(searcher.search({ query: 'copyright', params: { offset: 500, limit: 500 } })).toEqual([]); + searcher.closeIndex(); + searcher.loadIndex(dictionary); + }); +}); + +describe('query helpers', () => { + it('splits queries on the same boundaries as the index', () => { + expect(getQueryTerms('straße año')).toEqual(['straße', 'año']); + }); + + it('dedupes terms and keeps short ones', () => { + expect(getQueryTerms('Copyright (c) copyright MIT')).toEqual(['copyright', 'c', 'mit']); + }); + + it('skips verification only for a single term of up to 3 chars', () => { + expect(isExactIndexQuery(['sha'])).toBe(true); + expect(isExactIndexQuery(['go'])).toBe(true); + expect(isExactIndexQuery(['crypt'])).toBe(false); + expect(isExactIndexQuery(['sha', 'rsa'])).toBe(false); + }); + + it('verifies substrings case-insensitively', () => { + expect(containsAllTerms('SPDX-FileCopyrightText', ['copyright', 'text'])).toBe(true); + expect(containsAllTerms('copy the right way', ['copyright'])).toBe(false); + }); +}); + +describe('keyword index blacklist', () => { + const node = (nodePath: string, type = 'file') => ({ + getPath: () => nodePath, + getLabel: () => path.basename(nodePath), + getType: () => type, + }) as any; + + it('skips vendor, dot and notebook files unless all file types are included', () => { + const defaults = new BlackListKeyWordIndex(); + const all = new BlackListKeyWordIndex({ allExtensions: true }); + ['/vendor', '/.github', '/nb.ipynb'].forEach((p) => { + const type = p.includes('.ipynb') ? 'file' : 'folder'; + expect(defaults.evaluate(node(p, type))).toBe(true); + expect(all.evaluate(node(p, type))).toBe(false); + }); + }); + + it('skips every binary that the old extension list covered, in any case', () => { + const all = new BlackListKeyWordIndex({ allExtensions: true }); + ['.jpg', '.png', '.gif', '.woff', '.woff2', '.rar', '.jar', '.PNG', '.Jar'].forEach((ext) => { + expect(all.evaluate(node(`/assets/file${ext}`))).toBe(true); + }); + }); + + it('skips notebooks in any case unless all file types are included', () => { + expect(new BlackListKeyWordIndex().evaluate(node('/nb/Analysis.IPYNB'))).toBe(true); + expect(new BlackListKeyWordIndex({ allExtensions: true }).evaluate(node('/nb/Analysis.IPYNB'))).toBe(false); + }); + + it('keeps text files and folders named like binaries', () => { + const defaults = new BlackListKeyWordIndex(); + expect(defaults.evaluate(node('/src/main.c'))).toBe(false); + expect(defaults.evaluate(node('/license'))).toBe(false); + expect(defaults.evaluate(node('/icons.png', 'folder'))).toBe(false); + }); + + it('always skips binaries and never skips the root', () => { + const all = new BlackListKeyWordIndex({ allExtensions: true }); + expect(all.evaluate(node('/logo.png'))).toBe(true); + expect(all.evaluate(node('/lib.so'))).toBe(true); + expect(all.evaluate(node('', 'folder'))).toBe(false); + }); +}); diff --git a/src/__tests__/search/search-task.test.ts b/src/__tests__/search/search-task.test.ts new file mode 100644 index 000000000..324778485 --- /dev/null +++ b/src/__tests__/search/search-task.test.ts @@ -0,0 +1,138 @@ +/** + * @jest-environment node + */ +import fs from 'fs'; +import os from 'os'; +import path from 'path'; + +const project = { path: '', sourcePath: '' }; + +jest.mock('electron-log', () => ({ info: jest.fn(), warn: jest.fn(), error: jest.fn() })); +jest.mock('../../main/broadcastManager/BroadcastManager', () => ({ + broadcastManager: { get: () => ({ send: jest.fn() }) }, +})); +jest.mock('../../main/workspace/Workspace', () => ({ + workspace: { + getOpenProject: () => ({ getMyPath: () => project.path, getSourceCodePath: () => project.sourcePath }), + }, +})); +jest.mock('../../main/services/ProjectService', () => ({ + projectService: { getSourceCodeBasePath: () => project.sourcePath }, +})); +jest.mock('../../main/services/ModelProvider', () => ({ + modelProvider: { model: { file: { getAllBySearch: jest.fn() } } }, +})); + +/* eslint-disable import/first */ +import { Indexer } from '../../main/modules/searchEngine/indexer/Indexer'; +import { searcher } from '../../main/modules/searchEngine/searcher/Searcher'; +import { SearchTask } from '../../main/task/search/searchTask/SearchTask'; +import { modelProvider } from '../../main/services/ModelProvider'; +import * as utils from '../../main/utils/utils'; +/* eslint-enable import/first */ + +const FILES: Record = { + '/false-positive.txt': 'copyr yrig right', // every trigram of "copyright", but not the word + '/hit.txt': 'Copyright (c) 2020', + '/sha.txt': 'sha only', + '/go.txt': 'written in go, not in google', + '/google.txt': 'google only', +}; + +describe('SearchTask', () => { + let root: string; + const idToPath: Record = {}; + const getAllBySearch = modelProvider.model.file.getAllBySearch as jest.Mock; + + const run = (query: string, params: Record = { limit: 100 }) => new SearchTask().run({ query, params }); + const paths = (rows: Array<{ path: string }>) => rows.map((r) => r.path).sort(); + + beforeAll(async () => { + root = fs.mkdtempSync(path.join(os.tmpdir(), 'search-task-')); + const source = path.join(root, 'source'); + fs.mkdirSync(source); + const files = []; + const add = (file: string, content: string, id: number) => { + fs.writeFileSync(path.join(source, file), content); + idToPath[id] = file; + files.push({ fileId: id, path: file }); + }; + Object.entries(FILES).forEach(([file, content], i) => add(file, content, i + 1)); + // Odd files hold the word; even ones only its trigrams. + for (let i = 0; i < 30; i += 1) add(`/bulk${i}.txt`, i % 2 ? 'bulkword' : 'bulkw kword', 100 + i); + const indexer = new Indexer(); + await indexer.saveIndex(await indexer.index(files, source), path.join(root, 'dictionary')); + project.path = root; + }); + + beforeEach(() => { + project.sourcePath = path.join(root, 'source'); + searcher.closeIndex(); + // The real query joins one row per result, so every file comes back twice. + getAllBySearch.mockReset().mockImplementation(async (qb: any) => qb.builders[0].value.flatMap((id: number) => [ + { id, path: idToPath[id], usage: 'file' }, + { id, path: idToPath[id], usage: 'snippet' }, + ])); + }); + + afterAll(() => { + searcher.closeIndex(); + fs.rmSync(root, { recursive: true, force: true }); + }); + + it('drops trigram false positives', async () => { + expect(paths(await run('copyright'))).toEqual(['/hit.txt']); + }); + + it('returns unverified candidates when the source is unavailable', async () => { + project.sourcePath = ''; + expect(paths(await run('copyright'))).toEqual(['/false-positive.txt', '/hit.txt']); + }); + + it('returns a single trigram hit without reading files', async () => { + const read = jest.spyOn(utils, 'readTextFile'); + expect(paths(await run('sha'))).toEqual(['/sha.txt']); + expect(read).not.toHaveBeenCalled(); + await run('copyright'); + expect(read).toHaveBeenCalled(); + read.mockRestore(); + }); + + it('matches short keywords as whole words, alone or with longer terms', async () => { + expect(paths(await run('go'))).toEqual(['/go.txt']); + expect(paths(await run('go google'))).toEqual(['/go.txt']); + }); + + it('pages verified hits without verifying them again', async () => { + const pages = []; + pages.push(await run('bulkword', { limit: 5, offset: 0 })); + const calls = getAllBySearch.mock.calls.length; + for (let offset = 5; offset < 20; offset += 5) pages.push(await run('bulkword', { limit: 5, offset })); + const ids = pages.flat().map((r) => r.id); + expect(getAllBySearch.mock.calls.length).toBe(calls); + expect(ids.slice(0, 15)).toHaveLength(15); + expect(new Set(ids).size).toBe(15); + expect(ids.every((id) => id % 2 === 1)).toBe(true); + }); + + it('recomputes candidates on a new search or after the index is closed', async () => { + const spy = jest.spyOn(searcher, 'search'); + await run('bulkword', { limit: 5, offset: 0 }); + await run('bulkword', { limit: 5, offset: 5 }); + expect(spy).toHaveBeenCalledTimes(1); + await run('bulkword', { limit: 5, offset: 0 }); + expect(spy).toHaveBeenCalledTimes(2); + searcher.closeIndex(); + const page = await run('bulkword', { limit: 5, offset: 5 }); + expect(spy).toHaveBeenCalledTimes(3); + expect(page).toHaveLength(5); + spy.mockRestore(); + }); + + it('rejects when finished mid-run', async () => { + const task = new SearchTask(); + const pending = task.run({ query: 'bulkword', params: { limit: 5 } }); + task.finish(); + await expect(pending).rejects.toThrow('SearchTask is finished'); + }); +}); diff --git a/src/api/handlers/project.handler.ts b/src/api/handlers/project.handler.ts index 3da7c8e99..d7b500e16 100644 --- a/src/api/handlers/project.handler.ts +++ b/src/api/handlers/project.handler.ts @@ -31,6 +31,7 @@ api.handle(IpcChannels.PROJECT_OPEN_SCAN, async (event, payload: any) => { const p: Project = await workspace.openProject(new ProjectFilterPath(payload.path)); searcher.closeIndex(); + if (payload.mode !== ProjectAccessMode.READ_ONLY) projectService.rebuildOutdatedSearchIndex(p); // await projectService.lockProject(p.getProjectName(), mode); const response: ProjectOpenResponse = { logical_tree: p.getTree().getRootFolder(), diff --git a/src/api/handlers/search.handler.ts b/src/api/handlers/search.handler.ts index 22b4e3855..801d6edf8 100644 --- a/src/api/handlers/search.handler.ts +++ b/src/api/handlers/search.handler.ts @@ -4,21 +4,20 @@ import { SearchTask } from '../../main/task/search/searchTask/SearchTask'; import { ISearchTask } from '../../main/task/search/searchTask/ISearchTask'; import { ipcMain } from 'electron'; -let search = null; +let search: SearchTask = null; ipcMain.on(IpcChannels.SEARCH_ENGINE_SEARCH, async (event, params: ISearchTask) => { if (search) { search.finish(); } - search = new SearchTask(); - search - .run(params) - .then((response) => { - search = null; - event.sender.send(IpcChannels.SEARCH_ENGINE_SEARCH_RESPONSE, response); - return true; - }) - .catch((error: Error) => { - log.error('[SEARCH ENGINE]: ', error.message); - }); + try { + const task = new SearchTask(); + search = task; + const response = await task.run(params); + // A newer search may have started while this one was running. + if (search === task) search = null; + event.sender.send(IpcChannels.SEARCH_ENGINE_SEARCH_RESPONSE, response); + } catch (error: any) { + log.error('[SEARCH ENGINE]: ', error.message); + } }); diff --git a/src/main/migration/scripts/120.ts b/src/main/migration/scripts/120.ts index b7a497115..49f6f2d1f 100644 --- a/src/main/migration/scripts/120.ts +++ b/src/main/migration/scripts/120.ts @@ -33,16 +33,16 @@ async function indexMigration(projectPath: string) { QueryBuilderCreator.create(paths) ); const indexer = new Indexer(); - const filesToIndex = fileAdapter(files, metadata.scan_root); - const index = indexer.index(filesToIndex); + const filesToIndex = fileAdapter(files); + const index = await indexer.index(filesToIndex, metadata.scan_root); await indexer.saveIndex(index, `${projectPath}/dictionary/`); } } -function fileAdapter(modelFiles: any, scanRoot: string): Array { +function fileAdapter(modelFiles: any): Array { const filesToIndex = []; modelFiles.forEach((file: any) => { - filesToIndex.push({ fileId: file.id, path: `${scanRoot}${file.path}` }); + filesToIndex.push({ fileId: file.id, path: file.path }); }); return filesToIndex; } diff --git a/src/main/model/project/models/FileModel.ts b/src/main/model/project/models/FileModel.ts index cb781ce38..1ff6b5d7a 100644 --- a/src/main/model/project/models/FileModel.ts +++ b/src/main/model/project/models/FileModel.ts @@ -53,6 +53,11 @@ export class FileModel extends Model { return files; } + public async getAllPaths(): Promise> { + const call = promisify(this.connection.all.bind(this.connection)); + return call('SELECT fileId AS id, path FROM files;'); + } + public async getAllBySearch(queryBuilder?: QueryBuilder): Promise { const SQLQuery = this.getSQL(queryBuilder, queries.SQL_GET_ALL_FILES_BY_SEARCH, this.getEntityMapper()); const call:any = util.promisify(this.connection.all.bind(this.connection)); diff --git a/src/main/modules/searchEngine/indexer/Indexer.ts b/src/main/modules/searchEngine/indexer/Indexer.ts index 94377e63d..dacfc77ab 100644 --- a/src/main/modules/searchEngine/indexer/Indexer.ts +++ b/src/main/modules/searchEngine/indexer/Indexer.ts @@ -1,63 +1,50 @@ import fs from 'fs'; import log from 'electron-log'; import { getHeapStatistics } from 'node:v8'; +import path from 'path'; import { IIndexer } from './IIndexer'; import { IpcChannels } from '../../../../api/ipc-channels'; -import { getSearchConfig } from '../../../../shared/utils/search-utils'; +import { getSearchConfig, SEARCH_INDEX_VERSION, SEARCH_INDEX_VERSION_FILE } from '../../../../shared/utils/search-utils'; import { broadcastManager } from '../../../broadcastManager/BroadcastManager'; -import { workspace } from '../../../workspace/Workspace'; -import path from 'path'; -import { projectService } from '../../../services/ProjectService'; - +import { readTextFile } from '../../../utils/utils'; const { Index } = require('flexsearch'); +const BYTES_PER_MB = 1024 * 1024; + export class Indexer { private MAX_FILE_SIZE_MB = 100; private shouldStopIndexing(): boolean { const HEAP_BUFFER_MB = 200; - const MAX_HEAP_SIZE_MB = getHeapStatistics().heap_size_limit / (1024 * 1024); - const currentHeapMB = process.memoryUsage().heapUsed / (1024 * 1024); + const MAX_HEAP_SIZE_MB = getHeapStatistics().heap_size_limit / BYTES_PER_MB; + const currentHeapMB = process.memoryUsage().heapUsed / BYTES_PER_MB; return (MAX_HEAP_SIZE_MB - currentHeapMB) < HEAP_BUFFER_MB; } - private async getFileSizeMB(path: string): Promise { - try { - const stats = await fs.promises.stat(path); - return stats.size / (1024 * 1024); - } catch (e) { - console.error(`Error getting file size for ${path}:`, e); - return 0; - } - } - - public async index(files: Array) { - const sourceCodeBasePath = projectService.getSourceCodeBasePath() + public async index(files: Array, basePath: string) { const index = new Index(getSearchConfig()); for (let i = 0; i < files.length; i += 1) { + if (i % 100 === 0) { + this.sendToUI(IpcChannels.SCANNER_UPDATE_STATUS, { + processed: (i * 100) / files.length, + }); + } + if (this.shouldStopIndexing()) { + log.warn(`[ Indexer ]: heap limit reached, ${files.length - i} files were not indexed`); + break; + } try { - if (i % 100 === 0) { - this.sendToUI(IpcChannels.SCANNER_UPDATE_STATUS, { - processed: i * 100 / files.length, - }); - } - const absoluteFilePath = path.join(sourceCodeBasePath,files[i].path); - // Check file size first - const fileSizeMB = await this.getFileSizeMB(absoluteFilePath); + const absoluteFilePath = path.join(basePath, files[i].path); + const fileSizeMB = (await fs.promises.stat(absoluteFilePath)).size / BYTES_PER_MB; if (fileSizeMB > this.MAX_FILE_SIZE_MB) { - console.warn(`Skipping large file: ${files[i].path} (${fileSizeMB.toFixed(2)}MB)`); + log.warn(`[ Indexer ]: skipping large file ${files[i].path} (${fileSizeMB.toFixed(2)}MB)`); // eslint-disable-next-line no-continue continue; } - - if (this.shouldStopIndexing()) { - log.info('Skipping file indexing, maximum heap size exceeded'); - } else { - const fileContent = fs.readFileSync(absoluteFilePath, 'utf-8'); - index.add(files[i].fileId, fileContent); - } + const content = readTextFile(absoluteFilePath); + if (content !== null) index.add(files[i].fileId, content); } catch (e) { log.error(e); } @@ -66,15 +53,35 @@ export class Indexer { } public async saveIndex(index: any, pathToDictionary: string) { - if (fs.existsSync(pathToDictionary)) { - fs.rmSync(pathToDictionary, { recursive: true, force: true }); - } - fs.mkdirSync(pathToDictionary); + // Written aside and swapped in, so a failed or interrupted save never replaces the current dictionary. + const dictionaryPath = pathToDictionary.replace(/[\\/]+$/, ''); + this.removeOrphanTmpFolders(dictionaryPath); + const tmpPath = `${dictionaryPath}.${process.pid}-${Date.now()}.tmp`; + fs.mkdirSync(tmpPath); + const writes: Promise[] = []; await index.export((key: any, data: string | NodeJS.ArrayBufferView) => { - fs.writeFile(path.join(pathToDictionary, `${key}.json`), data !== undefined ? data : '', (err) => { - if (err) console.log(err); - }); + writes.push(fs.promises.writeFile(path.join(tmpPath, `${key}.json`), data !== undefined ? data : '')); }); + await Promise.all(writes); + await fs.promises.writeFile( + path.join(tmpPath, SEARCH_INDEX_VERSION_FILE), + JSON.stringify({ version: SEARCH_INDEX_VERSION }), + ); + fs.rmSync(pathToDictionary, { recursive: true, force: true }); + fs.renameSync(tmpPath, pathToDictionary); + } + + /** + * Removes temp folders left by a crashed save of a previous app run. This process's + * folders may belong to a save still in progress, so they are kept. + */ + private removeOrphanTmpFolders(dictionaryPath: string) { + const parent = path.dirname(dictionaryPath); + const prefix = `${path.basename(dictionaryPath)}.`; + const ownPrefix = `${prefix}${process.pid}-`; + fs.readdirSync(parent) + .filter((f) => f.startsWith(prefix) && f.endsWith('.tmp') && !f.startsWith(ownPrefix)) + .forEach((f) => fs.rmSync(path.join(parent, f), { recursive: true, force: true })); } private sendToUI(eventName, data: any) { diff --git a/src/main/modules/searchEngine/searcher/Searcher.ts b/src/main/modules/searchEngine/searcher/Searcher.ts index 92cc5c36a..e5d7bc25a 100644 --- a/src/main/modules/searchEngine/searcher/Searcher.ts +++ b/src/main/modules/searchEngine/searcher/Searcher.ts @@ -1,45 +1,105 @@ import fs from 'fs'; import path from 'path'; import { ISearcher } from './ISearcher'; -import { getSearchConfig } from '../../../../shared/utils/search-utils'; - +import { ISearchResult } from '../../../task/search/searchTask/ISearchResult'; +import { + getLegacySearchConfig, + getSearchConfig, + LEGACY_INDEX_VERSION, + SEARCH_INDEX_VERSION_FILE, +} from '../../../../shared/utils/search-utils'; const { Index } = require('flexsearch'); +const INDEX_IDLE_CLOSE_MS = 60000; + +/** + * Verification progress of one query, kept so later pages continue where the previous one stopped. + */ +export interface VerifiedSearch { + terms: string; + candidates: number[]; + cursor: number; + verified: ISearchResult[]; +} + +/** + * Returns the version of the dictionary stored at the given path. Missing dictionaries or version files read as legacy. + */ +export const readIndexVersion = (pathToDictionary: string): number => { + try { + const raw = fs.readFileSync(path.join(pathToDictionary, SEARCH_INDEX_VERSION_FILE), 'utf8'); + return JSON.parse(raw).version ?? LEGACY_INDEX_VERSION; + } catch (e) { + return LEGACY_INDEX_VERSION; + } +}; + class Searcher { private index: any; + private version: number; + + private closeTimer: NodeJS.Timeout | null; + + private verifiedSearch: VerifiedSearch | null; + constructor() { this.index = null; + this.version = LEGACY_INDEX_VERSION; + this.closeTimer = null; + this.verifiedSearch = null; } public search(params: ISearcher): number[] { if (this.index) { - const results = this.index.search(params.query, params.params ? params.params : null); - return results; + // flexsearch returns undefined when the offset is past the last hit. + return this.index.search(params.query, params.params ? params.params : null) ?? []; } return []; } + public getVersion(): number { + return this.version; + } + + /** + * Returns the cached verification of a query. It lives as long as the loaded index. + */ + public getVerifiedSearch(terms: string): VerifiedSearch | null { + return this.verifiedSearch?.terms === terms ? this.verifiedSearch : null; + } + + public setVerifiedSearch(verifiedSearch: VerifiedSearch) { + this.verifiedSearch = verifiedSearch; + } + public loadIndex(pathToDictionary: string) { if (!this.index) { - // @ts-ignore - const index = new Index(getSearchConfig()); if (fs.existsSync(pathToDictionary)) { - fs.readdirSync(pathToDictionary).forEach((file) => { - const filepath = path.join(pathToDictionary, file); - const filename = path.parse(file).name; - const data: any = fs.readFileSync(filepath, 'utf8'); - index.import(filename, data ?? null); - }); + this.version = readIndexVersion(pathToDictionary); + // @ts-ignore + const index = new Index(this.version === LEGACY_INDEX_VERSION ? getLegacySearchConfig() : getSearchConfig()); + fs.readdirSync(pathToDictionary) + .filter((file) => file !== SEARCH_INDEX_VERSION_FILE) + .forEach((file) => { + const filepath = path.join(pathToDictionary, file); + const filename = path.parse(file).name; + const data: any = fs.readFileSync(filepath, 'utf8'); + index.import(filename, data ?? null); + }); this.index = index; - setTimeout(this.closeIndex, 60000); // Close index after 1 minute + this.closeTimer = setTimeout(() => this.closeIndex(), INDEX_IDLE_CLOSE_MS); } } } public closeIndex() { + if (this.closeTimer) clearTimeout(this.closeTimer); + this.closeTimer = null; this.index = null; + this.version = LEGACY_INDEX_VERSION; + this.verifiedSearch = null; } } diff --git a/src/main/services/ProjectService.searchIndex.test.ts b/src/main/services/ProjectService.searchIndex.test.ts new file mode 100644 index 000000000..3898d8d99 --- /dev/null +++ b/src/main/services/ProjectService.searchIndex.test.ts @@ -0,0 +1,77 @@ +/** + * @jest-environment node + */ +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { Scanner } from '../task/scanner/types'; + +const mockSourcePath = { value: '' }; + +jest.mock('electron-log', () => ({ info: jest.fn(), warn: jest.fn(), error: jest.fn() })); +jest.mock('../workspace/Workspace', () => ({ + workspace: { getOpenProject: () => ({ getSourceCodePath: () => mockSourcePath.value }) }, +})); +jest.mock('../task/scanner/scannerPipelineFactory/ScannerPipelineFactory', () => ({ + ScannerPipelineFactory: { getScannerPipeline: jest.fn() }, +})); +jest.mock('./UserSettingService', () => ({ + userSettingService: { + get: jest.fn(() => ({ DEFAULT_WORKSPACE_INDEX: 0, WORKSPACES: [{ SCAN_SOURCES: '/', PATH: '/' }] })), + }, +})); +jest.mock('../task/search/indexTask/IndexTask', () => ({ IndexTask: jest.fn() })); + +/* eslint-disable import/first */ +import { IndexTask } from '../task/search/indexTask/IndexTask'; +import { projectService } from './ProjectService'; +/* eslint-enable import/first */ + +describe('ProjectService.rebuildOutdatedSearchIndex', () => { + let root: string; + const run = jest.fn(); + + const project = (stages = [Scanner.PipelineStage.CODE, Scanner.PipelineStage.SEARCH_INDEX]) => ({ + getMyPath: () => root, + getSourceCodePath: () => mockSourcePath.value, + metadata: { getScannerConfig: () => ({ pipelineStages: stages }) }, + }) as any; + + beforeEach(() => { + root = fs.mkdtempSync(path.join(os.tmpdir(), 'rebuild-index-')); + mockSourcePath.value = root; + run.mockReset().mockResolvedValue(true); + (IndexTask as unknown as jest.Mock).mockReset().mockImplementation(() => ({ run })); + }); + + afterEach(() => fs.rmSync(root, { recursive: true, force: true })); + + it('rebuilds a legacy dictionary when the source is available', async () => { + await projectService.rebuildOutdatedSearchIndex(project()); + expect(run).toHaveBeenCalledTimes(1); + }); + + it('skips current dictionaries, disabled index stages and missing sources', () => { + fs.mkdirSync(path.join(root, 'dictionary')); + fs.writeFileSync(path.join(root, 'dictionary', 'version.json'), JSON.stringify({ version: 2 })); + expect(projectService.rebuildOutdatedSearchIndex(project())).toBeNull(); + fs.rmSync(path.join(root, 'dictionary'), { recursive: true }); + + expect(projectService.rebuildOutdatedSearchIndex(project([Scanner.PipelineStage.CODE]))).toBeNull(); + + mockSourcePath.value = ''; + expect(projectService.rebuildOutdatedSearchIndex(project())).toBeNull(); + expect(run).not.toHaveBeenCalled(); + }); + + it('runs one rebuild at a time per project, also after a failure', async () => { + let fail: (e: Error) => void; + run.mockImplementationOnce(() => new Promise((_resolve, reject) => { fail = reject; })); + const first = projectService.rebuildOutdatedSearchIndex(project()); + expect(projectService.rebuildOutdatedSearchIndex(project())).toBeNull(); + fail(new Error('boom')); + await first; + await projectService.rebuildOutdatedSearchIndex(project()); + expect(run).toHaveBeenCalledTimes(2); + }); +}); diff --git a/src/main/services/ProjectService.ts b/src/main/services/ProjectService.ts index 54bd0ba23..b85fb17a0 100644 --- a/src/main/services/ProjectService.ts +++ b/src/main/services/ProjectService.ts @@ -1,4 +1,5 @@ import os from 'os'; +import fs from 'fs'; import * as util from 'util'; import log from 'electron-log'; import { ExtractFromProjectDTO, INewProject, IProject, Inventory, LOCK, ProjectAccessMode, ProjectKnowledgeExtractionResult, ProjectState, ReuseIdentificationTaskDTO } from '../../api/types'; @@ -14,8 +15,12 @@ import { ReuseIdentificationTask } from '../task/reuseIdentification/ReuseIdenti import { ReuseDependencyIdentificationTask } from '../task/reuseIdentification/ReuseDependencyIdentificationTask'; import ScannerMode = Scanner.ScannerMode; import path from 'path'; +import { IndexTask } from '../task/search/indexTask/IndexTask'; +import { readIndexVersion } from '../modules/searchEngine/searcher/Searcher'; +import { SEARCH_INDEX_FOLDER, SEARCH_INDEX_VERSION } from '../../shared/utils/search-utils'; class ProjectService { + private rebuildingSearchIndexes = new Set(); public async close(): Promise { const p = workspace.getOpenProject(); @@ -176,6 +181,29 @@ class ProjectService { return inventories; } + /** + * Rebuilds in background a keyword index created with an older config. Without source code the old one is kept. + * @returns The running rebuild, or null when none was started + */ + public rebuildOutdatedSearchIndex(p: Project): Promise | null { + try { + const stages = p.metadata.getScannerConfig()?.pipelineStages ?? []; + if (!stages.includes(Scanner.PipelineStage.SEARCH_INDEX)) return null; + if (readIndexVersion(path.join(p.getMyPath(), SEARCH_INDEX_FOLDER)) >= SEARCH_INDEX_VERSION) return null; + if (!p.getSourceCodePath() || !fs.existsSync(this.getSourceCodeBasePath())) return null; + if (this.rebuildingSearchIndexes.has(p.getMyPath())) return null; + this.rebuildingSearchIndexes.add(p.getMyPath()); + log.info('[ SEARCH INDEX ]: rebuilding outdated keyword index'); + return new IndexTask(p).run() + .then(() => undefined) + .catch((e) => log.error('[ SEARCH INDEX ]: rebuild failed', e)) + .finally(() => this.rebuildingSearchIndexes.delete(p.getMyPath())); + } catch (e) { + log.error('[ SEARCH INDEX ]: rebuild failed', e); + return null; + } + } + private getBaseSourcePath(){ const scanSources = userSettingService.get().WORKSPACES[userSettingService.get().DEFAULT_WORKSPACE_INDEX].SCAN_SOURCES; const workspacePath = userSettingService.get().WORKSPACES[userSettingService.get().DEFAULT_WORKSPACE_INDEX].PATH; diff --git a/src/main/task/search/indexTask/IndexTask.ts b/src/main/task/search/indexTask/IndexTask.ts index 539af4a9d..55f755020 100644 --- a/src/main/task/search/indexTask/IndexTask.ts +++ b/src/main/task/search/indexTask/IndexTask.ts @@ -1,16 +1,18 @@ import log from 'electron-log'; import { Scanner } from 'main/task/scanner/types'; import i18next from 'i18next'; +import path from 'path'; import { modelProvider } from '../../../services/ModelProvider'; +import { projectService } from '../../../services/ProjectService'; import { Indexer } from '../../../modules/searchEngine/indexer/Indexer'; import { IIndexer } from '../../../modules/searchEngine/indexer/IIndexer'; +import { searcher } from '../../../modules/searchEngine/searcher/Searcher'; import { workspace } from '../../../workspace/Workspace'; import { BlackListKeyWordIndex } from '../../../workspace/tree/blackList/BlackListKeyWordIndex'; -import { QueryBuilderCreator } from '../../../model/queryBuilder/QueryBuilderCreator'; import { Project } from '../../../workspace/Project'; import { ScannerStage } from '../../../../api/types'; -import path from 'path'; import { CollectFilesVisitor } from '../../../workspace/tree/visitor/CollectFilesVisitor'; +import { SEARCH_INDEX_FOLDER } from '../../../../shared/utils/search-utils'; export class IndexTask implements Scanner.IPipelineTask { private project: Project; @@ -29,30 +31,26 @@ export class IndexTask implements Scanner.IPipelineTask { public async run(): Promise { log.info('[ IndexTask init ]'); - const project = workspace.getOpenProject(); - if (!project) throw new Error('Not project opened'); - const collector = new CollectFilesVisitor(new BlackListKeyWordIndex()); + if (!workspace.getOpenProject()) throw new Error('Not project opened'); + const allExtensions = this.project.metadata.getScannerConfig()?.allExtensions ?? false; + const collector = new CollectFilesVisitor(new BlackListKeyWordIndex({ allExtensions })); this.project.getTree().getRootFolder().accept(collector); - const paths = collector.files.map((fi) => `'${fi.getPath()}'`).join(', '); + const paths = new Set(collector.files.map((fi) => fi.getPath())); - const files = await modelProvider.model.file.getAll( - QueryBuilderCreator.create({ paths }), - ); + const files = (await modelProvider.model.file.getAllPaths()).filter((f) => paths.has(f.path)); const projectPath = this.project.metadata.getMyPath(); - const dictionaryPath = path.join(projectPath, 'dictionary'); + const dictionaryPath = path.join(projectPath, SEARCH_INDEX_FOLDER); const indexer = new Indexer(); - const filesToIndex = this.fileAdapter(files); - const index = await indexer.index(filesToIndex); + const index = await indexer.index(this.fileAdapter(files), projectService.getSourceCodeBasePath()); await indexer.saveIndex(index, dictionaryPath); + // The project may have been closed while indexing; saving it would persist its released tree. + if (workspace.getOpenProject() !== this.project) return true; + searcher.closeIndex(); this.project.save(); return true; } - private fileAdapter(modelFiles: any): Array { - const filesToIndex = []; - modelFiles.forEach((file: any) => { - filesToIndex.push({ fileId: file.id, path: file.path }); - }); - return filesToIndex; + private fileAdapter(modelFiles: Array<{ id: number; path: string }>): Array { + return modelFiles.map((file) => ({ fileId: file.id, path: file.path })); } } diff --git a/src/main/task/search/searchTask/SearchTask.ts b/src/main/task/search/searchTask/SearchTask.ts index 897842ea8..f66b20022 100644 --- a/src/main/task/search/searchTask/SearchTask.ts +++ b/src/main/task/search/searchTask/SearchTask.ts @@ -1,34 +1,46 @@ +import fs from 'fs'; +import path from 'path'; import { searcher } from '../../../modules/searchEngine/searcher/Searcher'; import { workspace } from '../../../workspace/Workspace'; import { ITask } from '../../Task'; import { modelProvider } from '../../../services/ModelProvider'; +import { projectService } from '../../../services/ProjectService'; import { ISearchTask } from './ISearchTask'; import { QueryBuilderCreator } from '../../../model/queryBuilder/QueryBuilderCreator'; import { AppConfigDefault } from '../../../../config/AppConfigDefault'; import { ISearchResult } from './ISearchResult'; -import path from 'path'; +import { readTextFile } from '../../../utils/utils'; +import { + containsAllTerms, + getQueryTerms, + isExactIndexQuery, + SEARCH_INDEX_FOLDER, + SEARCH_INDEX_VERSION, +} from '../../../../shared/utils/search-utils'; + +const VERIFY_CHUNK_SIZE = 1000; + +// Files verified between yields to the event loop, so large searches don't freeze IPC. +const VERIFY_YIELD_EVERY = 50; export class SearchTask implements ITask> { private search = searcher; - private readonly DICTIONARY_FOLDER = 'dictionary'; - private isFinished: boolean; constructor() { - this.search.loadIndex(path.join(workspace.getOpenProject().getMyPath(), this.DICTIONARY_FOLDER)); + this.search.loadIndex(path.join(workspace.getOpenProject().getMyPath(), SEARCH_INDEX_FOLDER)); this.isFinished = false; } public async run(params: ISearchTask): Promise> { - if (!params.params?.limit || !params.params) { + if (!params.params?.limit) { const limit = AppConfigDefault.SEARCH_ENGINE_DEFAULT_LIMIT; - params.params = { limit }; + params.params = { ...params.params, limit }; } - const fileIds = this.search.search(params); - const results: Array = await modelProvider.model.file.getAllBySearch( - QueryBuilderCreator.create({ fileId: fileIds }) - ); + const results = this.search.getVersion() >= SEARCH_INDEX_VERSION + ? await this.searchVerified(params) + : await this.searchLegacy(params); const files = results.reduce((acc, curr) => { if (!acc[curr.path]) acc[curr.path] = curr; return acc; @@ -39,6 +51,63 @@ export class SearchTask implements ITask> { throw new Error('SearchTask is finished'); } + private async searchLegacy(params: ISearchTask): Promise> { + const fileIds = this.search.search(params); + return modelProvider.model.file.getAllBySearch(QueryBuilderCreator.create({ fileId: fileIds })); + } + + private async searchVerified(params: ISearchTask): Promise> { + const terms = getQueryTerms(params.query ?? ''); + if (terms.length === 0) return []; + + const offset = params.params.offset ?? 0; + const end = offset + params.params.limit; + const key = terms.join(' '); + + let state = offset === 0 ? null : this.search.getVerifiedSearch(key); + if (!state) { + const candidates = this.search.search({ query: terms.join(' '), params: { limit: Number.MAX_SAFE_INTEGER } }); + state = { terms: key, candidates, cursor: 0, verified: [] }; + this.search.setVerifiedSearch(state); + } + + const basePath = workspace.getOpenProject().getSourceCodePath() ? projectService.getSourceCodeBasePath() : null; + const canVerify = !isExactIndexQuery(terms) && basePath !== null && fs.existsSync(basePath); + if (!canVerify) { + return this.getFilesById(state.candidates.slice(offset, end)); + } + + while (state.verified.length < end && state.cursor < state.candidates.length) { + const chunk = state.candidates.slice(state.cursor, state.cursor + VERIFY_CHUNK_SIZE); + state.cursor += chunk.length; + // eslint-disable-next-line no-await-in-loop + const rows = await this.getFilesById(chunk); + for (let i = 0; i < rows.length && !this.isFinished; i += 1) { + // eslint-disable-next-line no-await-in-loop + if (i % VERIFY_YIELD_EVERY === 0) await new Promise((resolve) => { setImmediate(resolve); }); + try { + const content = readTextFile(path.join(basePath, rows[i].path)); + if (content !== null && containsAllTerms(content, terms)) state.verified.push(rows[i]); + } catch (e) { + // File removed from disk since it was indexed. + } + } + if (this.isFinished) break; + } + return state.verified.slice(offset, end); + } + + private async getFilesById(fileIds: number[]): Promise> { + if (fileIds.length === 0) return []; + const rows: Array = await modelProvider.model.file.getAllBySearch( + QueryBuilderCreator.create({ fileId: fileIds }), + ); + // A file joins one row per result; keep the first so pagination counts files. + const byId = new Map(); + rows.forEach((row) => { if (!byId.has(row.id)) byId.set(row.id, row); }); + return fileIds.filter((id) => byId.has(id)).map((id) => byId.get(id)); + } + public finish(): void { this.isFinished = true; } diff --git a/src/main/utils/utils.ts b/src/main/utils/utils.ts index 6eac8755a..40f5f33b2 100644 --- a/src/main/utils/utils.ts +++ b/src/main/utils/utils.ts @@ -11,6 +11,21 @@ export async function fileExists(filePath: string): Promise { } } +const BINARY_SNIFF_BYTES = 8000; + +/** + * Returns the text content of a file, or null for binaries. UTF-16 files are detected by their BOM. + */ +export function readTextFile(filePath: string): string | null { + const buffer = fs.readFileSync(filePath); + if (buffer.length >= 2 && buffer[0] === 0xff && buffer[1] === 0xfe) return buffer.toString('utf16le', 2); + if (buffer.length >= 2 && buffer[0] === 0xfe && buffer[1] === 0xff) { + return Buffer.from(buffer.subarray(2, buffer.length - (buffer.length % 2))).swap16().toString('utf16le'); + } + if (buffer.subarray(0, BINARY_SNIFF_BYTES).includes(0)) return null; + return buffer.toString('utf-8'); +} + export function toPosix(filePath: string): string { return filePath.replaceAll(path.sep, path.posix.sep); } diff --git a/src/main/workspace/ProjectZipper.ts b/src/main/workspace/ProjectZipper.ts index ed838eda8..eb33647e5 100644 --- a/src/main/workspace/ProjectZipper.ts +++ b/src/main/workspace/ProjectZipper.ts @@ -8,6 +8,7 @@ import { Project } from './Project'; import { workspace } from './Workspace'; import packageJson from '../../../release/app/package.json'; import AppConfig from '../../config/AppConfigModule'; +import { SEARCH_INDEX_FOLDER } from '../../shared/utils/search-utils'; const AdmZip = require('adm-zip'); @@ -48,7 +49,8 @@ export class ProjectZipper { // Before zipping, the api and api key needs to be removed from metadata. const dirContent = await fs.promises.readdir(projectPath); for (const file of dirContent) { - if (file !== 'metadata.json') { + // Skip dictionaries still being written by the indexer. + if (file !== 'metadata.json' && !(file.startsWith(`${SEARCH_INDEX_FOLDER}.`) && file.endsWith('.tmp'))) { if (file !== 'dictionary') { zip.addLocalFile(path.join(projectPath, file), path.basename(projectPath) + path.sep); } else { diff --git a/src/main/workspace/tree/blackList/BlackListKeyWordIndex.ts b/src/main/workspace/tree/blackList/BlackListKeyWordIndex.ts index 41dfc5c1e..490769f40 100644 --- a/src/main/workspace/tree/blackList/BlackListKeyWordIndex.ts +++ b/src/main/workspace/tree/blackList/BlackListKeyWordIndex.ts @@ -1,47 +1,37 @@ +import path from 'path'; import { BlackListAbstract } from './BlackListAbstract'; -import Node, { NodeStatus } from '../Node'; -import { workspace } from '../../Workspace'; +import Node from '../Node'; -import path from 'path'; +const isBinaryPath = require('is-binary-path'); -export class BlackListKeyWordIndex extends BlackListAbstract { - private scanRoot: string; +interface BlackListKeyWordIndexOptions { + // Mirrors the "Include all file types" scanner option. + allExtensions?: boolean; +} - private filesBlackList: Set; +export class BlackListKeyWordIndex extends BlackListAbstract { + private defaultSkippedFolders: Set; - private vendorFolders: Set; + private defaultSkippedExtensions: Set; - private extensions: Set; + private allExtensions: boolean; - constructor() { + constructor(options: BlackListKeyWordIndexOptions = {}) { super(); - this.filesBlackList = new Set([ - 'gradlew.bat', - 'mvnw', - 'mvnw.cmd', - 'gradle-wrapper.jar', - 'maven-wrapper.jar', - 'thumbs.db', - 'copying.lib', - ]); - - this.extensions = new Set([ - '.jpg', - '.png', - '.gif', - '.woff', - '.woff2', - '.rar', - '.jar', - '.ipynb', - ]); - - this.vendorFolders = new Set(['node_modules', 'vendor']); - - this.scanRoot = workspace.getOpenProject()?.getScanRoot(); + this.defaultSkippedFolders = new Set(['node_modules', 'vendor']); + // Notebooks are text, but their saved cell outputs bloat the index. + this.defaultSkippedExtensions = new Set(['.ipynb']); + this.allExtensions = options.allExtensions ?? false; } public evaluate(node: Node): boolean { - return node.getLabel().startsWith('.') || this.vendorFolders.has(node.getLabel()) || this.extensions.has(path.extname(node.getPath())); + // Root folder label is the project name, never filter it. + if (node.getPath() === '') return false; + const isFile = node.getType() === 'file'; + // Binaries hold no searchable text, so they are skipped even with "Include all file types". + if (isFile && isBinaryPath(node.getPath())) return true; + if (this.allExtensions) return false; + if (isFile && this.defaultSkippedExtensions.has(path.extname(node.getPath()).toLowerCase())) return true; + return node.getLabel().startsWith('.') || this.defaultSkippedFolders.has(node.getLabel()); } } diff --git a/src/shared/utils/search-utils.ts b/src/shared/utils/search-utils.ts index 272256535..0bfa24605 100644 --- a/src/shared/utils/search-utils.ts +++ b/src/shared/utils/search-utils.ts @@ -1,44 +1,70 @@ /** - * Return the default configuration for the search engine. It used by flexsearch on index and searcher creation. - * @returns The default configuration + * Version of the keyword search index. Bump it when the index config changes so + * dictionaries built with an older config are detected and rebuilt. */ -const getSearchConfig = (): Record => ({ - depth: 1, - bidirectional: 0, - resolution: 9, - minlength: 2, - stemmer: getDefaultStemmer(), -}); +const SEARCH_INDEX_VERSION = 2; + +// Dictionaries without a version file were built before versioning existed. +const LEGACY_INDEX_VERSION = 1; + +const SEARCH_INDEX_FOLDER = 'dictionary'; + +const SEARCH_INDEX_VERSION_FILE = 'version.json'; + +const NGRAM_SIZE = 3; + +// Caps the n-grams produced by very long tokens (minified code, base64 blobs). +const MAX_TOKEN_LENGTH = 256; /** - * Return the default stemmer for the search engine - * @param language The language to use for the stemmer + * Turns each token into its unique character trigrams, so a keyword matches anywhere + * inside a word ("crypt" in "encryption"). Tokens shorter than a trigram are kept whole. */ -const getDefaultStemmer = (language = 'US'): Record => { - return { - es: 'e', - ed: 'e', - ing: '', - }; +const toTrigrams = (tokens: string[]): string[] => { + const grams = new Set(); + tokens.forEach((token) => { + const t = token.length > MAX_TOKEN_LENGTH ? token.slice(0, MAX_TOKEN_LENGTH) : token; + if (t.length < NGRAM_SIZE) { + grams.add(t); + return; + } + for (let i = 0; i + NGRAM_SIZE <= t.length; i += 1) grams.add(t.substring(i, i + NGRAM_SIZE)); + }); + return Array.from(grams); }; /** - * Transform a query search in a list of tokens. This list will be enriched with the reverse of default stemmer. - * @param text The search query - * @return A list of tokens + * Flexsearch config shared by the indexer and the searcher. Dedupe and numeric are disabled: + * they collapse repeated letters and split numbers, which yields false positives. */ -const unStemmify = (text: string): string[] => { - const terms = getTerms(text); - const stemms = []; - terms.forEach(term => { - Object.keys(getDefaultStemmer()).forEach(key => { - if (term.endsWith(key)) { - stemms.push(term.replace(key, getDefaultStemmer()[key])); - } - }); - }); - return terms.concat(stemms); -}; +const getSearchConfig = (): Record => ({ + tokenize: 'strict', + resolution: 1, + fastupdate: false, + cache: false, + encoder: { + normalize: false, + dedupe: false, + numeric: false, + cache: false, + minlength: 1, + // The default drops long tokens entirely; toTrigrams truncates them instead. + maxlength: Number.MAX_SAFE_INTEGER, + prepare: (text: string) => text.toLowerCase(), + finalize: toTrigrams, + }, +}); + +/** + * Configuration of dictionaries built before SEARCH_INDEX_VERSION existed. + */ +const getLegacySearchConfig = (): Record => ({ + depth: 1, + bidirectional: 0, + resolution: 9, + minlength: 2, + stemmer: { es: 'e', ed: 'e', ing: '' }, +}); /** * Transform a query search in a list of crypto tokens. @@ -55,8 +81,39 @@ const unStemmifyCryptoKeywords = (text: string): string[] => { * @param querySearch The search query * @param regex The regex to split the query by. The default use same regex as the tokenizer */ -const getTerms = (querySearch: string, regex = /[\W_]+/): string[] => { +const getTerms = (querySearch: string, regex = /[^\p{L}\p{N}]+/u): string[] => { return querySearch.split(regex); }; -export { getSearchConfig, unStemmify, getTerms, unStemmifyCryptoKeywords }; +/** + * Returns the unique lowercase terms of a query. + */ +const getQueryTerms = (query: string): string[] => Array.from( + new Set(getTerms(query.toLowerCase()).filter((t) => t.length > 0)), +); + +/** + * True when index hits need no verification: a 3-char term is itself a trigram, and shorter + * terms are indexed whole, so they match whole words as before trigrams. + */ +const isExactIndexQuery = (terms: string[]): boolean => terms.length === 1 && terms[0].length <= NGRAM_SIZE; + +const containsAllTerms = (content: string, terms: string[]): boolean => { + const text = content.toLowerCase(); + return terms.every((t) => text.includes(t)); +}; + +export { + SEARCH_INDEX_VERSION, + LEGACY_INDEX_VERSION, + SEARCH_INDEX_FOLDER, + SEARCH_INDEX_VERSION_FILE, + NGRAM_SIZE, + getSearchConfig, + getLegacySearchConfig, + getTerms, + getQueryTerms, + isExactIndexQuery, + containsAllTerms, + unStemmifyCryptoKeywords, +};