Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
26 commits
Select commit Hold shift + click to select a range
1b536c7
Index vendor and dot folders in keyword search when all file types ar…
agustingroh Sep 29, 2026
c3a1e42
Match keyword search inside words using a trigram index
agustingroh Sep 29, 2026
0b3d82b
Skip in-progress search index folders on project export
agustingroh Sep 29, 2026
26c9964
Rebuild outdated keyword search indexes on project open
agustingroh Sep 29, 2026
ce7aa2d
Keep keyword search responses when searches overlap
agustingroh Sep 29, 2026
60dfdf2
Show minimum keyword length hint in keyword search
agustingroh Sep 29, 2026
a1000db
Update changelog for 1.41.2
agustingroh Sep 29, 2026
b71e64a
Bump version to 1.41.2-rc1
agustingroh Sep 29, 2026
a5ca28d
Return no results when legacy search pages past the last hit
agustingroh Sep 29, 2026
e52b1c2
Reset search index version when the index is closed
agustingroh Sep 29, 2026
3e40195
Split keyword queries on Unicode word boundaries
agustingroh Sep 29, 2026
5b726f4
Stop verifying files once a newer search starts
agustingroh Sep 29, 2026
53c85a7
Remove orphan search index temp folders
agustingroh Sep 29, 2026
ef5d9ed
Index notebooks when all file types are included
agustingroh Sep 29, 2026
16d6912
Move text file reading to shared utils
agustingroh Sep 29, 2026
bf8ff6b
Share keyword search index constants
agustingroh Sep 29, 2026
d05e24d
Keep keyword search paging state in the searcher
agustingroh Sep 29, 2026
13e1f3d
Move keyword index rebuild to ProjectService
agustingroh Sep 29, 2026
3633a6e
Search 1 and 2 character keywords as whole words
agustingroh Sep 29, 2026
c6c741e
Bump version to 1.41.2-rc2
agustingroh Sep 29, 2026
f7d151f
Move changelog entry to 1.41.1
agustingroh Sep 29, 2026
255d475
Set version to 1.41.1-rc1
agustingroh Sep 29, 2026
add1fbb
Pin macOS build runner to macOS 15
agustingroh Sep 29, 2026
33653c6
Rely on binary detection in the keyword index blacklist
agustingroh Sep 29, 2026
3be2641
Set changelog date for 1.41.1
agustingroh Sep 30, 2026
13e77c7
Bump version to 1.41.1
agustingroh Sep 30, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/_build-macos.yml
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@ on:
jobs:
build:
name: "Build for macOS"
runs-on: macos-latest
runs-on: macos-15

steps:
- name: Checkout repository
Expand Down
6 changes: 6 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,11 @@ All notable changes to this project will be documented in this file.
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).

## [1.41.1] - 2026-09-30
### Fixed
- Keyword search now finds the keyword anywhere in the text, including inside words (`copyright` in `SPDX-FileCopyrightText`). Existing projects are reindexed when opened if their source code is available.
- Keyword search now indexes `vendor/`, `node_modules/` and dot-folders when **Include all file types** is enabled.

## [1.41.0] - 2026-07-06
### Added
- **Import dependency identifications from another project.** The "Import identifications from…" flow now has an **Include dependencies** option that previews the source project's declared dependency identifications in a dedicated table and imports them alongside file/component identifications. Dependencies are matched by manifest path and PURL, and honor the same **Override previous work** toggle.
Expand Down Expand Up @@ -281,3 +286,4 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
[1.40.0]: https://github.com/scanoss/sbom-workbench/compare/v1.39.2...v1.40.0
[1.40.1]: https://github.com/scanoss/sbom-workbench/compare/v1.40.0...v1.40.1
[1.41.0]: https://github.com/scanoss/sbom-workbench/compare/v1.40.1...v1.41.0
[1.41.1]: https://github.com/scanoss/sbom-workbench/compare/v1.41.0...v1.41.1
2 changes: 1 addition & 1 deletion release/app/package.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "scanoss-workbench",
"version": "1.41.0",
"version": "1.41.1",
"description": "Desktop version to use SCANOSS OSS in your projects",
"license": "GPL-2.0-only",
"author": {
Expand Down
211 changes: 211 additions & 0 deletions src/__tests__/search/indexer-searcher.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,211 @@
/**
* @jest-environment node
*/
import fs from 'fs';
import os from 'os';
import path from 'path';
import { Indexer } from '../../main/modules/searchEngine/indexer/Indexer';
import { readIndexVersion, searcher } from '../../main/modules/searchEngine/searcher/Searcher';
import { BlackListKeyWordIndex } from '../../main/workspace/tree/blackList/BlackListKeyWordIndex';
import {
containsAllTerms,
getLegacySearchConfig,
getQueryTerms,
isExactIndexQuery,
SEARCH_INDEX_VERSION,
} from '../../shared/utils/search-utils';

jest.mock('electron-log', () => ({ info: jest.fn(), warn: jest.fn(), error: jest.fn() }));
jest.mock('../../main/broadcastManager/BroadcastManager', () => ({
broadcastManager: { get: () => ({ send: jest.fn() }) },
}));

const { Index } = require('flexsearch');

const utf16 = (text: string) => Buffer.concat([Buffer.from([0xff, 0xfe]), Buffer.from(text, 'utf16le')]);

const FIXTURES: Record<string, string | Buffer> = {
'/LICENSE': 'MIT License\n\nCopyright (c) 2020 Foo\n\nPermission is hereby granted',
'/src/main.c': 'int main() { return 0; } // Copyright (c) Bar',
'/src/header.ts': '// SPDX-FileCopyrightText: 2024 Acme\nexport const x = 1;',
'/src/crypto.js': 'const cipher = createEncryption(key);',
'/src/hash.go': 'sum := sha256.Sum256(data)',
'/docs/utf16.txt': utf16('hello utf16 world'),
'/img/logo.png': Buffer.from([0x89, 0x50, 0x4e, 0x47, 0x00, 0x00, 0x1a, 0x0a]),
"/src/it's.txt": 'quoted path content',
'/docs/intl.md': 'Diseño del año, straße und 日本語テキスト',
};

describe('keyword search index', () => {
let root: string;
let source: string;
let dictionary: string;
const ids: Record<string, number> = {};

const search = (query: string, params: Record<string, number> = { limit: 1000 }) => (
searcher.search({ query, params }).sort((a, b) => a - b)
);

beforeAll(async () => {
root = fs.mkdtempSync(path.join(os.tmpdir(), 'search-index-'));
source = path.join(root, 'source');
dictionary = path.join(root, 'dictionary');
const files = Object.entries(FIXTURES).map(([file, content], i) => {
fs.mkdirSync(path.join(source, path.dirname(file)), { recursive: true });
fs.writeFileSync(path.join(source, file), content);
ids[file] = i + 1;
return { fileId: i + 1, path: file };
});
for (let i = 0; i < 600; i += 1) {
const file = `/bulk/file${i}.txt`;
fs.mkdirSync(path.join(source, 'bulk'), { recursive: true });
fs.writeFileSync(path.join(source, file), `bulk content number ${i}`);
files.push({ fileId: 1000 + i, path: file });
}
const indexer = new Indexer();
await indexer.saveIndex(await indexer.index(files, source), dictionary);
searcher.closeIndex();
searcher.loadIndex(dictionary);
});

afterAll(() => {
searcher.closeIndex();
fs.rmSync(root, { recursive: true, force: true });
});

it('writes the index version', () => {
expect(readIndexVersion(dictionary)).toBe(SEARCH_INDEX_VERSION);
expect(searcher.getVersion()).toBe(SEARCH_INDEX_VERSION);
});

it('finds keywords in extension-less files and inline comments', () => {
expect(search('copyright')).toEqual([ids['/LICENSE'], ids['/src/main.c'], ids['/src/header.ts']]);
});

it('finds 1 and 2 char keywords as whole words', () => {
expect(search('c')).toEqual([ids['/LICENSE'], ids['/src/main.c']]);
expect(search('x')).toEqual([ids['/src/header.ts']]);
});

it('finds keywords inside words', () => {
expect(search('crypt')).toEqual([ids['/src/crypto.js']]);
expect(search('sha')).toEqual([ids['/src/hash.go']]);
});

it('indexes UTF-16 files and skips binaries', () => {
expect(search('utf16')).toEqual([ids['/docs/utf16.txt']]);
expect(search('png')).toEqual([]);
});

it('indexes paths containing quotes', () => {
expect(search('quoted')).toEqual([ids["/src/it's.txt"]]);
});

it('finds non-ASCII keywords', () => {
expect(search(getQueryTerms('año').join(' '))).toEqual([ids['/docs/intl.md']]);
expect(search(getQueryTerms('日本語').join(' '))).toEqual([ids['/docs/intl.md']]);
});

it('removes temp folders left by a crashed save', async () => {
const other = path.join(root, 'other');
fs.mkdirSync(other);
const orphan = path.join(other, 'dictionary.1-1.tmp');
fs.mkdirSync(orphan);
const indexer = new Indexer();
await indexer.saveIndex(await indexer.index([{ fileId: 1, path: '/LICENSE' }], source), path.join(other, 'dictionary'));
expect(fs.readdirSync(other)).toEqual(['dictionary']);
});

it('requires every term of a multi-term query', () => {
expect(search('copyright permission')).toEqual([ids['/LICENSE']]);
});

it('pages results with offset', () => {
const first = search('bulk', { limit: 500, offset: 0 });
const second = search('bulk', { limit: 500, offset: 500 });
expect(first).toHaveLength(500);
expect(second).toHaveLength(100);
expect(first.filter((id) => second.includes(id))).toEqual([]);
});

it('loads a legacy dictionary without a version file', () => {
const legacy = path.join(root, 'legacy');
fs.mkdirSync(legacy);
const index = new Index(getLegacySearchConfig());
index.add(1, 'legacy copyright notice');
index.export((key: string, data: string) => fs.writeFileSync(path.join(legacy, `${key}.json`), data ?? ''));
searcher.closeIndex();
searcher.loadIndex(legacy);
expect(searcher.getVersion()).toBe(1);
expect(searcher.search({ query: 'copyright' })).toEqual([1]);
expect(searcher.search({ query: 'copyright', params: { offset: 500, limit: 500 } })).toEqual([]);
searcher.closeIndex();
searcher.loadIndex(dictionary);
});
});

describe('query helpers', () => {
it('splits queries on the same boundaries as the index', () => {
expect(getQueryTerms('straße año')).toEqual(['straße', 'año']);
});

it('dedupes terms and keeps short ones', () => {
expect(getQueryTerms('Copyright (c) copyright MIT')).toEqual(['copyright', 'c', 'mit']);
});

it('skips verification only for a single term of up to 3 chars', () => {
expect(isExactIndexQuery(['sha'])).toBe(true);
expect(isExactIndexQuery(['go'])).toBe(true);
expect(isExactIndexQuery(['crypt'])).toBe(false);
expect(isExactIndexQuery(['sha', 'rsa'])).toBe(false);
});

it('verifies substrings case-insensitively', () => {
expect(containsAllTerms('SPDX-FileCopyrightText', ['copyright', 'text'])).toBe(true);
expect(containsAllTerms('copy the right way', ['copyright'])).toBe(false);
});
});

describe('keyword index blacklist', () => {
const node = (nodePath: string, type = 'file') => ({
getPath: () => nodePath,
getLabel: () => path.basename(nodePath),
getType: () => type,
}) as any;

it('skips vendor, dot and notebook files unless all file types are included', () => {
const defaults = new BlackListKeyWordIndex();
const all = new BlackListKeyWordIndex({ allExtensions: true });
['/vendor', '/.github', '/nb.ipynb'].forEach((p) => {
const type = p.includes('.ipynb') ? 'file' : 'folder';
expect(defaults.evaluate(node(p, type))).toBe(true);
expect(all.evaluate(node(p, type))).toBe(false);
});
});

it('skips every binary that the old extension list covered, in any case', () => {
const all = new BlackListKeyWordIndex({ allExtensions: true });
['.jpg', '.png', '.gif', '.woff', '.woff2', '.rar', '.jar', '.PNG', '.Jar'].forEach((ext) => {
expect(all.evaluate(node(`/assets/file${ext}`))).toBe(true);
});
});

it('skips notebooks in any case unless all file types are included', () => {
expect(new BlackListKeyWordIndex().evaluate(node('/nb/Analysis.IPYNB'))).toBe(true);
expect(new BlackListKeyWordIndex({ allExtensions: true }).evaluate(node('/nb/Analysis.IPYNB'))).toBe(false);
});

it('keeps text files and folders named like binaries', () => {
const defaults = new BlackListKeyWordIndex();
expect(defaults.evaluate(node('/src/main.c'))).toBe(false);
expect(defaults.evaluate(node('/license'))).toBe(false);
expect(defaults.evaluate(node('/icons.png', 'folder'))).toBe(false);
});

it('always skips binaries and never skips the root', () => {
const all = new BlackListKeyWordIndex({ allExtensions: true });
expect(all.evaluate(node('/logo.png'))).toBe(true);
expect(all.evaluate(node('/lib.so'))).toBe(true);
expect(all.evaluate(node('', 'folder'))).toBe(false);
});
});
138 changes: 138 additions & 0 deletions src/__tests__/search/search-task.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,138 @@
/**
* @jest-environment node
*/
import fs from 'fs';
import os from 'os';
import path from 'path';

const project = { path: '', sourcePath: '' };

jest.mock('electron-log', () => ({ info: jest.fn(), warn: jest.fn(), error: jest.fn() }));
jest.mock('../../main/broadcastManager/BroadcastManager', () => ({
broadcastManager: { get: () => ({ send: jest.fn() }) },
}));
jest.mock('../../main/workspace/Workspace', () => ({
workspace: {
getOpenProject: () => ({ getMyPath: () => project.path, getSourceCodePath: () => project.sourcePath }),
},
}));
jest.mock('../../main/services/ProjectService', () => ({
projectService: { getSourceCodeBasePath: () => project.sourcePath },
}));
jest.mock('../../main/services/ModelProvider', () => ({
modelProvider: { model: { file: { getAllBySearch: jest.fn() } } },
}));

/* eslint-disable import/first */
import { Indexer } from '../../main/modules/searchEngine/indexer/Indexer';
import { searcher } from '../../main/modules/searchEngine/searcher/Searcher';
import { SearchTask } from '../../main/task/search/searchTask/SearchTask';
import { modelProvider } from '../../main/services/ModelProvider';
import * as utils from '../../main/utils/utils';
/* eslint-enable import/first */

const FILES: Record<string, string> = {
'/false-positive.txt': 'copyr yrig right', // every trigram of "copyright", but not the word
'/hit.txt': 'Copyright (c) 2020',
'/sha.txt': 'sha only',
'/go.txt': 'written in go, not in google',
'/google.txt': 'google only',
};

describe('SearchTask', () => {
let root: string;
const idToPath: Record<number, string> = {};
const getAllBySearch = modelProvider.model.file.getAllBySearch as jest.Mock;

const run = (query: string, params: Record<string, number> = { limit: 100 }) => new SearchTask().run({ query, params });
const paths = (rows: Array<{ path: string }>) => rows.map((r) => r.path).sort();

beforeAll(async () => {
root = fs.mkdtempSync(path.join(os.tmpdir(), 'search-task-'));
const source = path.join(root, 'source');
fs.mkdirSync(source);
const files = [];
const add = (file: string, content: string, id: number) => {
fs.writeFileSync(path.join(source, file), content);
idToPath[id] = file;
files.push({ fileId: id, path: file });
};
Object.entries(FILES).forEach(([file, content], i) => add(file, content, i + 1));
// Odd files hold the word; even ones only its trigrams.
for (let i = 0; i < 30; i += 1) add(`/bulk${i}.txt`, i % 2 ? 'bulkword' : 'bulkw kword', 100 + i);
const indexer = new Indexer();
await indexer.saveIndex(await indexer.index(files, source), path.join(root, 'dictionary'));
project.path = root;
});

beforeEach(() => {
project.sourcePath = path.join(root, 'source');
searcher.closeIndex();
// The real query joins one row per result, so every file comes back twice.
getAllBySearch.mockReset().mockImplementation(async (qb: any) => qb.builders[0].value.flatMap((id: number) => [
{ id, path: idToPath[id], usage: 'file' },
{ id, path: idToPath[id], usage: 'snippet' },
]));
});

afterAll(() => {
searcher.closeIndex();
fs.rmSync(root, { recursive: true, force: true });
});

it('drops trigram false positives', async () => {
expect(paths(await run('copyright'))).toEqual(['/hit.txt']);
});

it('returns unverified candidates when the source is unavailable', async () => {
project.sourcePath = '';
expect(paths(await run('copyright'))).toEqual(['/false-positive.txt', '/hit.txt']);
});

it('returns a single trigram hit without reading files', async () => {
const read = jest.spyOn(utils, 'readTextFile');
expect(paths(await run('sha'))).toEqual(['/sha.txt']);
expect(read).not.toHaveBeenCalled();
await run('copyright');
expect(read).toHaveBeenCalled();
read.mockRestore();
});

it('matches short keywords as whole words, alone or with longer terms', async () => {
expect(paths(await run('go'))).toEqual(['/go.txt']);
expect(paths(await run('go google'))).toEqual(['/go.txt']);
});

it('pages verified hits without verifying them again', async () => {
const pages = [];
pages.push(await run('bulkword', { limit: 5, offset: 0 }));
const calls = getAllBySearch.mock.calls.length;
for (let offset = 5; offset < 20; offset += 5) pages.push(await run('bulkword', { limit: 5, offset }));
const ids = pages.flat().map((r) => r.id);
expect(getAllBySearch.mock.calls.length).toBe(calls);
expect(ids.slice(0, 15)).toHaveLength(15);
expect(new Set(ids).size).toBe(15);
expect(ids.every((id) => id % 2 === 1)).toBe(true);
});

it('recomputes candidates on a new search or after the index is closed', async () => {
const spy = jest.spyOn(searcher, 'search');
await run('bulkword', { limit: 5, offset: 0 });
await run('bulkword', { limit: 5, offset: 5 });
expect(spy).toHaveBeenCalledTimes(1);
await run('bulkword', { limit: 5, offset: 0 });
expect(spy).toHaveBeenCalledTimes(2);
searcher.closeIndex();
const page = await run('bulkword', { limit: 5, offset: 5 });
expect(spy).toHaveBeenCalledTimes(3);
expect(page).toHaveLength(5);
spy.mockRestore();
});

it('rejects when finished mid-run', async () => {
const task = new SearchTask();
const pending = task.run({ query: 'bulkword', params: { limit: 5 } });
task.finish();
await expect(pending).rejects.toThrow('SearchTask is finished');
});
});
1 change: 1 addition & 0 deletions src/api/handlers/project.handler.ts
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,7 @@ api.handle(IpcChannels.PROJECT_OPEN_SCAN, async (event, payload: any) => {

const p: Project = await workspace.openProject(new ProjectFilterPath(payload.path));
searcher.closeIndex();
if (payload.mode !== ProjectAccessMode.READ_ONLY) projectService.rebuildOutdatedSearchIndex(p);
// await projectService.lockProject(p.getProjectName(), mode);
const response: ProjectOpenResponse = {
logical_tree: p.getTree().getRootFolder(),
Expand Down
Loading
Loading