diff --git a/packages/app-vscode/package.json b/packages/app-vscode/package.json index 460168c8f2..a29de507c3 100644 --- a/packages/app-vscode/package.json +++ b/packages/app-vscode/package.json @@ -334,27 +334,6 @@ "order": 2, "description": "Keep a local, sanitized command history. This history is never sent to our servers, and any commands that may contain text will be sanitized. These statistics can be used in the future for doing local analyses to determine ways you can improve your Cursorless efficiency. We may also support a way for you to send your statistics to us for analysis in the future, but this will be opt-in only." }, - "cursorless.tokenHatSplittingMode.preserveCase": { - "type": "boolean", - "default": false, - "markdownDescription": "Whether to distinguish between uppercase and lower case letters for hats. Set this to `true` if you have separate terms for uppercase letters in your `` capture." - }, - "cursorless.tokenHatSplittingMode.lettersToPreserve": { - "type": "array", - "items": { - "type": "string" - }, - "default": [], - "description": "A list of characters whose accents should not be stripped. This can be used, for example, if you would like to strip all accents except for those of a few characters, which you would add to this list." - }, - "cursorless.tokenHatSplittingMode.symbolsToPreserve": { - "type": "array", - "items": { - "type": "string" - }, - "default": [], - "markdownDescription": "A list of symbols that shouldn't be normalized by the token hat splitter. Add any extra symbols here that you have added to your `` capture. Unlike the Accents To Preserve setting, these symbols won't even undergo case normalisation, so you would need separate terms for the lowercase and uppercase versions (if the symbol has a notion of upper and lower case)." - }, "cursorless.decorationDebounceDelayMs": { "type": "number", "default": 50, diff --git a/packages/app-vscode/src/extension.ts b/packages/app-vscode/src/extension.ts index 26a39e2d10..64d71d5ec6 100644 --- a/packages/app-vscode/src/extension.ts +++ b/packages/app-vscode/src/extension.ts @@ -4,6 +4,7 @@ import { FakeCommandServerApi, FakeIDE, NormalizedIDE, + showWarning, } from "@cursorless/lib-common"; import type { EngineProps } from "@cursorless/lib-engine"; import { CommandHistory, createCursorlessEngine } from "@cursorless/lib-engine"; @@ -26,6 +27,7 @@ import { createScopeVisualizer } from "./createScopeVisualizer"; import { createTreeSitter } from "./createTreeSitter"; import { createTutorial } from "./createTutorial"; import { createVscodeIde } from "./createVscodeIde"; +import type { VscodeIDE } from "./ide/vscode/VscodeIDE"; import { InstallationDependencies } from "./InstallationDependencies"; import { KeyboardCommands } from "./keyboard/KeyboardCommands"; import { registerCommands } from "./registerCommands"; @@ -153,6 +155,8 @@ export async function activate( hats, ); + deprecatedSettings(vscodeIDE); + registerCommands( context, vscodeIDE, @@ -190,3 +194,17 @@ export async function activate( : undefined, }; } + +function deprecatedSettings(ide: VscodeIDE) { + // DEPRECATED @ 2026-10-05 + const value = vscodeApi.workspace + .getConfiguration("cursorless") + .get("tokenHatSplittingMode"); + if (value != null) { + void showWarning( + ide.messages, + "tokenHatSplittingModeDeprecated", + "The 'cursorless.tokenHatSplittingMode' setting is deprecated and not needed when using an up to date Cursorless Talon", + ); + } +} diff --git a/packages/app-vscode/src/registerCommands.ts b/packages/app-vscode/src/registerCommands.ts index 0f9299b79b..06d4b63895 100644 --- a/packages/app-vscode/src/registerCommands.ts +++ b/packages/app-vscode/src/registerCommands.ts @@ -3,14 +3,12 @@ import type { CommandHistoryStorage, CursorlessCommandId, ScopeType, + TalonSpokenForms, } from "@cursorless/lib-common"; import { CURSORLESS_COMMAND_ID } from "@cursorless/lib-common"; import type { CommandApi, StoredTargetMap } from "@cursorless/lib-engine"; import { analyzeCommandHistory } from "@cursorless/lib-engine"; -import type { - CheatSheetCommandArg, - FileSystemTalonSpokenForms, -} from "@cursorless/lib-node-common"; +import type { CheatSheetCommandArg } from "@cursorless/lib-node-common"; import { showCheatsheet } from "@cursorless/lib-node-common"; import type { ScopeTestRecorder, @@ -45,7 +43,7 @@ export function registerCommands( tutorial: VscodeTutorial, installationDependencies: InstallationDependencies, storedTargets: StoredTargetMap, - talonSpokenForms: FileSystemTalonSpokenForms, + talonSpokenForms: TalonSpokenForms, ): void { const runCommandWrapper = async (run: () => Promise) => { try { diff --git a/packages/app-web-docs/src/docs/user/alphabet-and-symbols.md b/packages/app-web-docs/src/docs/user/alphabet-and-symbols.md index 5ca134958e..f8e1766606 100644 --- a/packages/app-web-docs/src/docs/user/alphabet-and-symbols.md +++ b/packages/app-web-docs/src/docs/user/alphabet-and-symbols.md @@ -7,6 +7,8 @@ sidebar_position: 7 Cursorless uses the [Talon Community](https://github.com/talonhub/community) alphabet, digits, and symbol spoken forms via the [`user.any_alphanumeric_key`](https://github.com/talonhub/community/blob/607c3415f5f29a5f75db6fe5648e37f514f62ac5/core/keys/keys.py#L71-L74) capture. +Additional characters provided by the lists in this capture are automatically preserved when allocating hats. See [Unicode support](unicode-support.md) for case handling, accented letters, and custom symbols. + ## Alphabet | Character | Default spoken form | diff --git a/packages/app-web-docs/src/docs/user/unicode-support.md b/packages/app-web-docs/src/docs/user/unicode-support.md index 39f526768e..bfdd83df4c 100644 --- a/packages/app-web-docs/src/docs/user/unicode-support.md +++ b/packages/app-web-docs/src/docs/user/unicode-support.md @@ -5,31 +5,42 @@ sidebar_position: 2 # Unicode support -Cursorless has first-class support for Unicode. By default, when constructing hats, Cursorless will ignore capitalization and any accents or diacritics over letters. For example, each of the following four tokens could each be selected by saying `"take air"` if there were a gray hat over their first letter (note the accents on the first letter for some of them): +Cursorless has first-class support for Unicode. With the default Talon alphabet, Cursorless ignores capitalization and accents or diacritics when constructing hats. For example, each of the following four tokens could be selected by saying `"take air"` if there were a gray hat over its first letter: - africa - áfrica - Africa - África -For Unicode symbols that are not letters, and that are not speakable by default, for example emoji, Chinese characters, etc, we have a special "character" called `"special"` that can be used. So for example, if there were a blue hat over a '😄' character, you could say `"take blue special"` to select it. As always, the spoken form `"special"` can be [customized](customization.md). +Characters that do not have a known spoken form, even after normalization, can be referred to using `"special"`. For example, if there were a blue hat over a `😄` character, you could say `"take blue special"` to select it. This also works for letters that cannot be normalized to a known character, such as Chinese characters. As always, the spoken form `"special"` can be [customized](customization.md). ## Advanced customization -The above setup will allow you to refer to any Unicode token, and is sufficient for most users. However, if you have overridden your `` capture to contain characters other than lowercase letters and the default symbols, you can tell Cursorless to be less aggressive with its normalization, so that it can allocate hats more efficiently. Note that this is not necessary in order to refer to these tokens; it just makes hat allocation slightly more efficient. +With an up-to-date Cursorless Talon installation, Cursorless automatically preserves additional characters provided by the Talon lists used in your `` capture. Add your spoken forms there; no editor setting is needed. This lets Cursorless allocate hats separately for these characters instead of grouping them with normalized letters or `"special"`. + +The old `cursorless.tokenHatSplittingMode` settings (`preserveCase`, `lettersToPreserve`, and `symbolsToPreserve`) are deprecated and can be removed from your editor settings. ### Preserving case -If you have a separate alphabet for uppercase letters as part of ``, you can enable the _Cursorless › Token Hat Splitting Mode: **Preserve Case**_ setting, and Cursorless will distinguish between lower and uppercase letters. +If your `` capture provides separate spoken forms for uppercase letters, Cursorless preserves those letters automatically. For example, if `"upper air"` produces `A`, a gray hat on the `A` in `Africa` can be addressed with `"take upper air"`, while a gray hat on lowercase `a` uses `"take air"`. ### Preserving special letters -If you have terms in `` for letters with accents, such as `é`, or other letters, such as `ø` or `ꝏ`, you can use the following setting: +If your capture provides terms for accented letters, such as `é`, or other letters, such as `ø` or `ꝏ`, Cursorless preserves them automatically. + +For example, if `"a umlaut"` produces `ä` and you have no separate form for `Ä`, a gray hat over the first letter of either `ällo` or `Ällo` can be addressed with `"take a umlaut"`. If you also provide a spoken form for `Ä`, Cursorless treats it separately. Providing only `Ä` does not preserve lowercase `ä`; lowercase `ä` still normalizes to `a`. + +### Preserving symbols + +Symbols provided by your capture are also preserved automatically. For example, if `"sigma"` produces `σ` and `"upper sigma"` produces `Σ`, a blue hat on those characters can be addressed with `"take blue sigma"` and `"take blue upper sigma"`, respectively. -#### _Cursorless › Token Hat Splitting Mode: **Letters To Preserve**_ +## Normalization order -Add any accented letters to this list that you have a spoken form for in ``. Cursorless will then preserve their accents during normalization. Note that Cursorless will still do case normalisation for these letters if you have [case preservation](#preserving-case) on. So, for example, if the list contains `ä`, and you'd like to refer to the token `Ällo` with a hat over the first letter (`Ä`), you can use your spoken form for `ä`. +Cursorless first normalizes each character to Unicode NFC so that equivalent representations, such as an accented letter written as one codepoint or with a combining mark, are treated the same. It then: -### _Cursorless › Token Hat Splitting Mode: **Symbols To Preserve**_ +1. Preserves the character if it is a default character or is provided by Talon. +2. Otherwise, converts it to lowercase and uses that form if it is known. +3. Otherwise, strips accents and diacritics and uses the resulting form if it is known. +4. Otherwise, assigns it to `"special"`. -Any Unicode symbols in this list will not undergo any normalisation, even case normalisation. Use this list for symbols for which you have spoken forms in `` that shouldn't be normalised at all, even by case. For example, if you have spoken forms for `Σ` and `σ`, and would like Cursorless not to treat them the same, you can add them to this list. +If custom characters cannot be loaded from Talon, Cursorless uses the default alphabet, digits, and symbols. diff --git a/packages/lib-common/src/FakeTalonSpokenForms.ts b/packages/lib-common/src/FakeTalonSpokenForms.ts new file mode 100644 index 0000000000..abf9c31d24 --- /dev/null +++ b/packages/lib-common/src/FakeTalonSpokenForms.ts @@ -0,0 +1,34 @@ +import type { + SpokenFormEntry, + TalonSpokenForms, + TalonSpokenFormsPayload, +} from "@cursorless/lib-common"; + +export class FakeTalonSpokenForms implements TalonSpokenForms { + public static fromGraphemes(graphemes: string[]): FakeTalonSpokenForms { + return new FakeTalonSpokenForms( + graphemes.map((grapheme) => ({ + type: "grapheme", + id: grapheme, + spokenForms: [], + })), + ); + } + + constructor(private spokenForms: SpokenFormEntry[]) {} + + getSpokenForms(): Promise { + return Promise.resolve({ + version: -1, + spokenForms: this.spokenForms, + }); + } + + onDidChange() { + return { + dispose: () => { + // No-op + }, + }; + } +} diff --git a/packages/lib-common/src/ide/types/Configuration.ts b/packages/lib-common/src/ide/types/Configuration.ts index 67c1ee6751..9799b51f06 100644 --- a/packages/lib-common/src/ide/types/Configuration.ts +++ b/packages/lib-common/src/ide/types/Configuration.ts @@ -4,7 +4,6 @@ import type { Disposable } from "./ide.types"; import type { GetFieldType, Paths } from "./Paths"; export type CursorlessConfiguration = { - tokenHatSplittingMode: TokenHatSplittingMode; wordSeparators: string[]; experimental: { hatStability: HatStability; @@ -19,11 +18,6 @@ export type CursorlessConfigKey = keyof CursorlessConfiguration; export type ConfigurationScope = { languageId: string }; export const CONFIGURATION_DEFAULTS: CursorlessConfiguration = { - tokenHatSplittingMode: { - preserveCase: false, - lettersToPreserve: [], - symbolsToPreserve: [], - }, wordSeparators: ["_"], decorationDebounceDelayMs: 50, experimental: { @@ -51,24 +45,3 @@ export interface Configuration { onDidChangeConfiguration(listener: Listener): Disposable; } - -export interface TokenHatSplittingMode { - /** - * Whether to distinguished between uppercase and lower case letters for hat - */ - preserveCase: boolean; - - /** - * A list of characters whose accents should not be stripped. This can be - * used, for example, if you would like to strip all accents except for those - * of a few characters, which you can add to this string. - */ - lettersToPreserve: string[]; - - /** - * A list of symbols that shouldn't be normalized by the token hat splitter. - * Add any extra symbols here that you have added to your - * capture. - */ - symbolsToPreserve: string[]; -} diff --git a/packages/lib-common/src/index.ts b/packages/lib-common/src/index.ts index 0afcf8578f..2e0a86281d 100644 --- a/packages/lib-common/src/index.ts +++ b/packages/lib-common/src/index.ts @@ -6,6 +6,7 @@ export * from "./Debouncer"; export * from "./errors"; export * from "./extensionDependencies"; export * from "./FakeCommandServerApi"; +export * from "./FakeTalonSpokenForms"; export * from "./ide/fake/FakeIDE"; export * from "./ide/inMemoryTextEditor/InMemoryTextDocument"; export * from "./ide/inMemoryTextEditor/InMemoryTextEditor"; diff --git a/packages/lib-common/src/types/HatTokenMap.ts b/packages/lib-common/src/types/HatTokenMap.ts index aff7ff2c2d..27542812ab 100644 --- a/packages/lib-common/src/types/HatTokenMap.ts +++ b/packages/lib-common/src/types/HatTokenMap.ts @@ -7,10 +7,7 @@ import type { Token } from "./Token"; * Maps from (hatStyle, character) pairs to tokens */ export interface HatTokenMap { - allocateHats( - forceTokenHats?: TokenHat[], - options?: HatAllocationOptions, - ): Promise; + allocateHats(options?: HatAllocationOptions): Promise; getReadableMap(usePrePhraseSnapshot: boolean): Promise; } @@ -20,6 +17,11 @@ export interface HatAllocationOptions { * Defaults to `false`. Forced hats still apply when starting fresh. */ startFresh?: boolean; + + /** If supplied, force the allocator to use these hats + * for the given tokens. This is used for the tutorial, and for testing. + */ + forceTokenHats?: TokenHat[]; } export interface TokenHat { diff --git a/packages/lib-engine/src/core/HatAllocator.test.ts b/packages/lib-engine/src/core/HatAllocator.test.ts index c284f1d953..511ffa78ef 100644 --- a/packages/lib-engine/src/core/HatAllocator.test.ts +++ b/packages/lib-engine/src/core/HatAllocator.test.ts @@ -8,6 +8,7 @@ import { Range, tokenHatToPlainObject, } from "@cursorless/lib-common"; +import { DisabledTalonSpokenForms } from "../disabledComponents/DisabledTalonSpokenForms"; import { TokenGraphemeSplitter } from "../tokenGraphemeSplitter"; import { HatAllocator } from "./HatAllocator"; import { IndividualHatMap } from "./IndividualHatMap"; @@ -16,11 +17,12 @@ import { RangeUpdater } from "./updateSelections/RangeUpdater"; suite("HatAllocator", () => { test("initial allocation is independent of earlier hat assignments", async () => { const ide = new HatTestIDE(); + const talonSpokenForms = new DisabledTalonSpokenForms(); ide.configuration.mockConfiguration("experimental", { ...ide.configuration.getOwnConfiguration("experimental"), hatStability: HatStability.stable, }); - const splitter = new TokenGraphemeSplitter(ide); + const splitter = new TokenGraphemeSplitter(ide, talonSpokenForms); const rangeUpdater = new RangeUpdater(ide); const map = new IndividualHatMap(ide, splitter, rangeUpdater); const hats: Hats = { @@ -66,17 +68,18 @@ suite("HatAllocator", () => { try { // Establish the expected allocation with no earlier editor events. - await allocator.allocateHats([forcedWorld]); + await allocator.allocateHats({ forceTokenHats: [forcedWorld] }); const expected = snapshot(); // Simulate an allocation that occurred before fixture initialization. - await allocator.allocateHats([previousHello]); - await allocator.allocateHats([forcedWorld]); + await allocator.allocateHats({ forceTokenHats: [previousHello] }); + await allocator.allocateHats({ forceTokenHats: [forcedWorld] }); assert.notDeepEqual(snapshot(), expected); assert.ok(map.getToken("blue", "h"), "Normal allocation preserves hats"); - await allocator.allocateHats([forcedWorld], { + await allocator.allocateHats({ startFresh: true, + forceTokenHats: [forcedWorld], }); assert.deepEqual(snapshot(), expected); assert.ok(map.getToken("default", "w"), "Forced hats are still applied"); diff --git a/packages/lib-engine/src/core/HatAllocator.ts b/packages/lib-engine/src/core/HatAllocator.ts index 456358ff63..ecb53ea176 100644 --- a/packages/lib-engine/src/core/HatAllocator.ts +++ b/packages/lib-engine/src/core/HatAllocator.ts @@ -3,7 +3,6 @@ import type { HatAllocationOptions, Hats, IDE, - TokenHat, } from "@cursorless/lib-common"; import type { TokenGraphemeSplitter } from "../tokenGraphemeSplitter"; import { allocateHats } from "../util/allocateHats"; @@ -16,6 +15,7 @@ interface Context { export class HatAllocator { private disposables: Disposable[] = []; + private startFresh = false; constructor( private ide: IDE, @@ -25,9 +25,10 @@ export class HatAllocator { ) { ide.disposeOnExit(this); - const debouncer = new DecorationDebouncer(ide.configuration, () => - this.allocateHats(), - ); + const debouncer = new DecorationDebouncer(ide.configuration, () => { + this.allocateHats({ startFresh: this.startFresh }); + this.startFresh = false; + }); this.disposables.push( this.hats.onDidChangeEnabledHatStyles(debouncer.run), @@ -47,9 +48,12 @@ export class HatAllocator { ide.onDidChangeTextEditorSelection(debouncer.run), // An Event which fires when the visible ranges of an editor has changed. ide.onDidChangeTextEditorVisibleRanges(debouncer.run), - // Re-draw hats on grapheme splitting algorithm change in case they - // changed their token hat splitting setting. - tokenGraphemeSplitter.registerAlgorithmChangeListener(debouncer.run), + // Re-draw hats on grapheme splitting algorithm change. + tokenGraphemeSplitter.registerAlgorithmChangeListener(() => { + // When the grapheme splitting algorithm changes, we need to start fresh to ensure hats are correctly allocated. + this.startFresh = true; + debouncer.run(); + }), debouncer, ); @@ -58,14 +62,12 @@ export class HatAllocator { /** * Allocate hats to the visible tokens. * - * @param forceTokenHats If supplied, force the allocator to use these hats - * for the given tokens. This is used for the tutorial, and for testing. * @param options Controls whether to start fresh without previous hat assignments. */ - async allocateHats( - forceTokenHats?: TokenHat[], - { startFresh = false }: HatAllocationOptions = {}, - ) { + async allocateHats({ + startFresh = false, + forceTokenHats, + }: HatAllocationOptions = {}) { const activeMap = await this.context.getActiveMap(); // Forced graphemes won't have been normalized diff --git a/packages/lib-engine/src/core/HatTokenMapImpl.ts b/packages/lib-engine/src/core/HatTokenMapImpl.ts index a06299ad04..80472ec0ef 100644 --- a/packages/lib-engine/src/core/HatTokenMapImpl.ts +++ b/packages/lib-engine/src/core/HatTokenMapImpl.ts @@ -5,7 +5,6 @@ import type { Hats, IDE, ReadOnlyHatMap, - TokenHat, } from "@cursorless/lib-common"; import type { TokenGraphemeSplitter } from "../tokenGraphemeSplitter"; import type { Debug } from "./Debug"; @@ -64,12 +63,9 @@ export class HatTokenMapImpl implements HatTokenMap { /** * Allocate hats to the visible tokens. - * - * @param forceTokenHats If supplied, force the allocator to use these hats - * for the given tokens. This is used for the tutorial, and for testing. */ - allocateHats(forceTokenHats?: TokenHat[], options?: HatAllocationOptions) { - return this.hatAllocator.allocateHats(forceTokenHats, options); + allocateHats(options?: HatAllocationOptions) { + return this.hatAllocator.allocateHats(options); } private async getActiveMap() { diff --git a/packages/lib-engine/src/cursorlessEngine.ts b/packages/lib-engine/src/cursorlessEngine.ts index 086965e09c..07b21a578a 100644 --- a/packages/lib-engine/src/cursorlessEngine.ts +++ b/packages/lib-engine/src/cursorlessEngine.ts @@ -61,7 +61,11 @@ export async function createCursorlessEngine({ const debug = new Debug(injectedIde); const rangeUpdater = new RangeUpdater(injectedIde); - const tokenGraphemeSplitter = new TokenGraphemeSplitter(injectedIde); + const tokenGraphemeSplitter = new TokenGraphemeSplitter( + injectedIde, + talonSpokenForms, + ); + await tokenGraphemeSplitter.ready; const storedTargets = new StoredTargetMap(); const keyboardTargetUpdater = new KeyboardTargetUpdater( diff --git a/packages/lib-engine/src/scripts/transformRecordedTests/checkMarks.ts b/packages/lib-engine/src/scripts/transformRecordedTests/checkMarks.ts index 0ca33da06d..6ac2f9a130 100644 --- a/packages/lib-engine/src/scripts/transformRecordedTests/checkMarks.ts +++ b/packages/lib-engine/src/scripts/transformRecordedTests/checkMarks.ts @@ -2,6 +2,7 @@ import assert from "node:assert/strict"; import { uniq } from "lodash-es"; import type { TestCaseFixtureLegacy } from "@cursorless/lib-common"; import { FakeIDE } from "@cursorless/lib-common"; +import { DisabledTalonSpokenForms } from "../../disabledComponents/DisabledTalonSpokenForms"; import { extractTargetKeys } from "../../testUtil/extractTargetKeys"; import { TokenGraphemeSplitter } from "../../tokenGraphemeSplitter/tokenGraphemeSplitter"; import { getPartialTargetDescriptors } from "../../util/getPartialTargetDescriptors"; @@ -10,7 +11,10 @@ import { canonicalize } from "./transformations/canonicalize"; export function checkMarks(originalFixture: TestCaseFixtureLegacy): undefined { const command = canonicalize(originalFixture).command; - const graphemeSplitter = new TokenGraphemeSplitter(new FakeIDE()); + const graphemeSplitter = new TokenGraphemeSplitter( + new FakeIDE(), + new DisabledTalonSpokenForms(), + ); const targetedMarks = getPartialTargetDescriptors(command.action).flatMap( extractTargetKeys, diff --git a/packages/lib-engine/src/spokenForms/CustomSpokenForms.ts b/packages/lib-engine/src/spokenForms/CustomSpokenForms.ts index a4b68deaf7..ede9e66bbe 100644 --- a/packages/lib-engine/src/spokenForms/CustomSpokenForms.ts +++ b/packages/lib-engine/src/spokenForms/CustomSpokenForms.ts @@ -36,6 +36,7 @@ type Writable = { export class CustomSpokenForms { private disposable: Disposable; private notifier = new Notifier(); + private updateGeneration = 0; /** * A promise that resolves when the custom spoken forms have been loaded. @@ -77,6 +78,7 @@ export class CustomSpokenForms { onDidChangeCustomSpokenForms = this.notifier.registerListener; private async updateSpokenFormMaps(): Promise { + const generation = ++this.updateGeneration; let allCustomEntries: SpokenFormEntry[]; let spokenFormsVersion: number; @@ -86,24 +88,32 @@ export class CustomSpokenForms { try { const payload = await this.talonSpokenForms.getSpokenForms(); + + if (generation !== this.updateGeneration) { + // Another update has occurred since this one started, so we should abort. + return; + } + allCustomEntries = payload.spokenForms; spokenFormsVersion = payload.version; if (allCustomEntries.length === 0) { throw new Error("Custom spoken forms list empty"); } } catch (error) { + if (generation !== this.updateGeneration) { + return; + } if (error instanceof NeedsInitialTalonUpdateError) { // Handle case where spokenForms.json doesn't exist yet this.needsInitialTalonUpdate_ = true; } else if (error instanceof DisabledCustomSpokenFormsError) { // Do nothing: this ide doesn't currently support custom spoken forms } else { - console.error("Error loading custom spoken forms", error); const msg = getErrorMessage(error).replace(/\.$/u, ""); void showError( this.ide.messages, "CustomSpokenForms.updateSpokenFormMaps", - `Error loading custom spoken forms: ${msg}. Falling back to default spoken forms.`, + `Error loading custom spoken forms: ${msg}. Falling back to default.`, ); } diff --git a/packages/lib-engine/src/test/talonSpokenFormsUpdates.test.ts b/packages/lib-engine/src/test/talonSpokenFormsUpdates.test.ts new file mode 100644 index 0000000000..fab6214751 --- /dev/null +++ b/packages/lib-engine/src/test/talonSpokenFormsUpdates.test.ts @@ -0,0 +1,148 @@ +import assert from "node:assert/strict"; +import type { + IDE, + TalonSpokenForms, + TalonSpokenFormsPayload, +} from "@cursorless/lib-common"; +import { FakeIDE, NeedsInitialTalonUpdateError } from "@cursorless/lib-common"; +import { CustomSpokenForms } from "../spokenForms/CustomSpokenForms"; +import { defaultSpokenFormMap } from "../spokenForms/defaultSpokenFormMap"; +import { TokenGraphemeSplitter } from "../tokenGraphemeSplitter"; + +const consumers = [ + { + name: "CustomSpokenForms", + create: (ide: IDE, forms: TalonSpokenForms) => { + const consumer = new CustomSpokenForms(ide, forms); + return { + ready: consumer.customSpokenFormsInitialized, + subscribe: consumer.onDidChangeCustomSpokenForms, + assertNewerResult: (failed: boolean, notifications: number) => { + if (failed) { + assert.deepEqual(consumer.spokenFormMap, defaultSpokenFormMap); + } else { + assert.deepEqual( + consumer.spokenFormMap.grapheme["ä"]?.spokenForms, + ["ä"], + ); + } + assert.equal(consumer.needsInitialTalonUpdate, failed); + assert.equal(notifications, 1); + }, + snapshot: () => ({ + map: consumer.spokenFormMap, + needsInitialTalonUpdate: consumer.needsInitialTalonUpdate, + }), + dispose: () => consumer.dispose(), + }; + }, + }, + { + name: "TokenGraphemeSplitter", + create: (ide: IDE, forms: TalonSpokenForms) => { + const consumer = new TokenGraphemeSplitter(ide, forms); + return { + ready: consumer.ready, + subscribe: consumer.registerAlgorithmChangeListener, + assertNewerResult: (failed: boolean, notifications: number) => { + assert.equal(consumer.normalizeGrapheme("ä"), failed ? "a" : "ä"); + assert.equal(notifications, failed ? 0 : 1); + }, + snapshot: () => consumer.normalizeGrapheme("ä"), + dispose: () => consumer.dispose(), + }; + }, + }, +]; + +function payload(id: string): TalonSpokenFormsPayload { + return { + version: 1, + spokenForms: [{ type: "grapheme", id, spokenForms: [id] }], + }; +} + +function createRequest() { + let resolve!: (payload: TalonSpokenFormsPayload) => void; + let reject!: (error: Error) => void; + // oxlint-disable-next-line promise/param-names + const promise = new Promise((res, rej) => { + resolve = res; + reject = rej; + }); + return { promise, resolve, reject }; +} + +for (const { name, create } of consumers) { + suite(`${name} overlapping updates`, () => { + for (const outcome of ["success", "missing file", "invalid JSON"]) { + for (const newerFails of [false, true]) { + test(`ignores older ${outcome} after newer ${newerFails ? "failure" : "success"}`, async () => { + const requests: ReturnType[] = []; + // oxlint-disable-next-line unicorn/consistent-function-scoping + let update: () => void | Promise = () => { + // No-op + }; + const forms: TalonSpokenForms = { + getSpokenForms: () => { + const request = createRequest(); + requests.push(request); + return request.promise; + }, + onDidChange: (listener) => { + update = listener; + return { + dispose: () => { + // No-op + }, + }; + }, + }; + const ide = new FakeIDE(); + let messages = 0; + ide.messages.showMessage = () => { + messages++; + return Promise.resolve(undefined); + }; + const consumer = create(ide, forms); + + try { + requests[0].resolve(payload("a")); + await consumer.ready; + let notifications = 0; + consumer.subscribe(() => { + notifications++; + }); + + const older = update(); + const newer = update(); + if (newerFails) { + requests[2].reject(new NeedsInitialTalonUpdateError("Missing")); + } else { + requests[2].resolve(payload("ä")); + } + await newer; + consumer.assertNewerResult(newerFails, notifications); + const expected = consumer.snapshot(); + + if (outcome === "success") { + requests[1].resolve(payload("ø")); + } else if (outcome === "missing file") { + requests[1].reject(new NeedsInitialTalonUpdateError("Missing")); + } else { + requests[1].reject(new SyntaxError("Invalid JSON")); + } + await older; + + assert.deepEqual(consumer.snapshot(), expected); + consumer.assertNewerResult(newerFails, notifications); + assert.equal(messages, 0); + } finally { + consumer.dispose(); + ide.exit(); + } + }); + } + } + }); +} diff --git a/packages/lib-engine/src/test/unicode.test.ts b/packages/lib-engine/src/test/unicode.test.ts index 881a3f97e5..ca31a11650 100644 --- a/packages/lib-engine/src/test/unicode.test.ts +++ b/packages/lib-engine/src/test/unicode.test.ts @@ -8,6 +8,7 @@ import type { } from "@cursorless/lib-common"; import { FakeIDE, + FakeTalonSpokenForms, InMemoryTextEditor, Notifier, Range, @@ -33,14 +34,10 @@ suite("Unicode support", () => { for (const { name, content, grapheme } of unicodeTestCases) { test(`targets ${name}`, async () => { const ide = new UnicodeTestIDE(content); - ide.configuration.mockConfiguration("tokenHatSplittingMode", { - preserveCase: false, - lettersToPreserve, - symbolsToPreserve: [], - }); const engine = await createCursorlessEngine({ ide, hats: createTestHats(), + talonSpokenForms: FakeTalonSpokenForms.fromGraphemes(lettersToPreserve), }); const tokenRange = new Range(0, 0, 0, content.length); const forcedHat: TokenHat = { @@ -56,8 +53,9 @@ suite("Unicode support", () => { }; try { - await engine.hatTokenMap.allocateHats([forcedHat], { + await engine.hatTokenMap.allocateHats({ startFresh: true, + forceTokenHats: [forcedHat], }); const hatMap = await engine.hatTokenMap.getReadableMap(false); diff --git a/packages/lib-engine/src/tokenGraphemeSplitter/tokenGraphemeSplitter.test.ts b/packages/lib-engine/src/tokenGraphemeSplitter/tokenGraphemeSplitter.test.ts index 9f66b50b95..ff55de98c2 100644 --- a/packages/lib-engine/src/tokenGraphemeSplitter/tokenGraphemeSplitter.test.ts +++ b/packages/lib-engine/src/tokenGraphemeSplitter/tokenGraphemeSplitter.test.ts @@ -1,7 +1,11 @@ // oxlint-disable no-inline-comments import assert from "node:assert/strict"; -import type { TokenHatSplittingMode } from "@cursorless/lib-common"; -import { FakeIDE } from "@cursorless/lib-common"; +import type { TalonSpokenForms } from "@cursorless/lib-common"; +import { + FakeIDE, + FakeTalonSpokenForms, + Notifier, +} from "@cursorless/lib-common"; import { TokenGraphemeSplitter, UNKNOWN } from "./tokenGraphemeSplitter"; /** @@ -18,7 +22,7 @@ type CompactGrapheme = [string, number, number]; type TestCase = [string, CompactGrapheme[]]; interface SplittingModeTestCases { - tokenHatSplittingMode: Partial; + graphemes: string[]; extraTestCases: TestCase[]; } @@ -36,7 +40,7 @@ const commonTestCases: TestCase[] = [ const tests: SplittingModeTestCases[] = [ { - tokenHatSplittingMode: {}, + graphemes: [], extraTestCases: [ [ "\u00F1", // ñ as single codepoint @@ -65,9 +69,7 @@ const tests: SplittingModeTestCases[] = [ ], }, { - tokenHatSplittingMode: { - preserveCase: true, - }, + graphemes: [], extraTestCases: [ [ "\u00F1", // ñ as single codepoint @@ -79,30 +81,28 @@ const tests: SplittingModeTestCases[] = [ ], [ "\u00D1", // Ñ as single codepoint - [["N", 0, 1]], + [["n", 0, 1]], ], [ "\u004E\u0303", // Ñ using combining mark - [["N", 0, 2]], + [["n", 0, 2]], ], ["ꝏ", [[UNKNOWN, 0, 1]]], ["ø", [["o", 0, 1]]], ["æ", [[UNKNOWN, 0, 1]]], ["Ꝏ", [[UNKNOWN, 0, 1]]], - ["Ø", [["O", 0, 1]]], + ["Ø", [["o", 0, 1]]], ["Æ", [[UNKNOWN, 0, 1]]], ], }, { - tokenHatSplittingMode: { - lettersToPreserve: [ - "\u00E4", // ä, NFC-normalised - "\u00E5", // å, NFC-normalised - "ꝏ", - "ø", - "æ", - ], - }, + graphemes: [ + "\u00E4", // ä, NFC-normalised + "\u00E5", // å, NFC-normalised + "ꝏ", + "ø", + "æ", + ], extraTestCases: [ [ "\u00F1", // ñ as single codepoint @@ -157,12 +157,10 @@ const tests: SplittingModeTestCases[] = [ ], }, { - tokenHatSplittingMode: { - lettersToPreserve: [ - "\u0061\u0308", // ä, NFD-normalised - "\u0061\u030A", // å, NFD-normalised - ], - }, + graphemes: [ + "\u0061\u0308", // ä, NFD-normalised + "\u0061\u030A", // å, NFD-normalised + ], extraTestCases: [ [ "\u00E4\u00E5", // äå, NFC-normalised @@ -181,16 +179,13 @@ const tests: SplittingModeTestCases[] = [ ], }, { - tokenHatSplittingMode: { - preserveCase: true, - lettersToPreserve: [ - "\u00E4", // ä, NFC-normalised - "\u00E5", // å, NFC-normalised - "ꝏ", - "ø", - "æ", - ], - }, + graphemes: [ + "\u00E4", // ä, NFC-normalised + "\u00E5", // å, NFC-normalised + "ꝏ", + "ø", + "æ", + ], extraTestCases: [ [ "\u00E4\u00E5", // äå, NFC-normalised @@ -202,55 +197,51 @@ const tests: SplittingModeTestCases[] = [ [ "\u00C4\u00C5", // ÄÅ, NFC-normalised [ - ["\u00C4", 0, 1], // Ä, NFC-normalised - ["\u00C5", 1, 2], // Å, NFC-normalised + ["\u00E4", 0, 1], // ä, NFC-normalised + ["\u00E5", 1, 2], // å, NFC-normalised ], ], ["ꝏ", [["ꝏ", 0, 1]]], ["ø", [["ø", 0, 1]]], ["æ", [["æ", 0, 1]]], - ["Ꝏ", [["Ꝏ", 0, 1]]], - ["Ø", [["Ø", 0, 1]]], - ["Æ", [["Æ", 0, 1]]], + ["Ꝏ", [["ꝏ", 0, 1]]], + ["Ø", [["ø", 0, 1]]], + ["Æ", [["æ", 0, 1]]], ], }, { - tokenHatSplittingMode: { - lettersToPreserve: [ - "\u00C4", // Ä, NFC-normalised - "\u00C5", // Å, NFC-normalised - "Ꝏ", - "Ø", - "Æ", - ], - }, + graphemes: [ + "\u00C4", // Ä, NFC- + "\u00C5", // Å, NFC-normalised + "Ꝏ", + "Ø", + "Æ", + ], extraTestCases: [ [ "\u00E4\u00E5", // äå, NFC-normalised [ - ["\u00E4", 0, 1], // ä, NFC-normalised - ["\u00E5", 1, 2], // å, NFC-normalised + ["a", 0, 1], + ["a", 1, 2], ], ], [ "\u00C4\u00C5", // ÄÅ, NFC-normalised [ - ["\u00E4", 0, 1], // ä, NFC-normalised - ["\u00E5", 1, 2], // å, NFC-normalised + ["\u00C4", 0, 1], // Ä, NFC-normalised + ["\u00C5", 1, 2], // Å, NFC-normalised ], ], - ["ꝏ", [["ꝏ", 0, 1]]], - ["ø", [["ø", 0, 1]]], - ["æ", [["æ", 0, 1]]], - ["Ꝏ", [["ꝏ", 0, 1]]], - ["Ø", [["ø", 0, 1]]], - ["Æ", [["æ", 0, 1]]], + ["ꝏ", [[UNKNOWN, 0, 1]]], + ["ø", [["o", 0, 1]]], + ["æ", [[UNKNOWN, 0, 1]]], + ["Ꝏ", [["Ꝏ", 0, 1]]], + ["Ø", [["Ø", 0, 1]]], + ["Æ", [["Æ", 0, 1]]], ], }, { - tokenHatSplittingMode: { - symbolsToPreserve: ["🙃", "Σ", "σ"], - }, + graphemes: ["🙃", "Σ", "σ"], extraTestCases: [ [ "\u00F1", // ñ as single codepoint @@ -274,21 +265,10 @@ const tests: SplittingModeTestCases[] = [ }, ]; -const tokenHatSplittingDefaults: TokenHatSplittingMode = { - preserveCase: false, - lettersToPreserve: [], - symbolsToPreserve: [], -}; - -for (const { tokenHatSplittingMode, extraTestCases } of tests) { - suite(`getTokenGraphemes(${JSON.stringify(tokenHatSplittingMode)})`, () => { +for (const { graphemes, extraTestCases } of tests) { + suite(`getTokenGraphemes(${JSON.stringify(graphemes)})`, () => { const ide = new FakeIDE(); - - ide.configuration.mockConfiguration("tokenHatSplittingMode", { - ...tokenHatSplittingDefaults, - ...tokenHatSplittingMode, - }); - + const talonSpokenForms = FakeTalonSpokenForms.fromGraphemes(graphemes); const testCases = [...commonTestCases, ...extraTestCases]; for (const [input, compactExpectedOutput] of testCases) { @@ -302,12 +282,112 @@ for (const { tokenHatSplittingMode, extraTestCases } of tests) { const displayOutput = expectedOutput.map(({ text }) => text).join(", "); - test(`${input} -> ${displayOutput}`, () => { - const actualOutput = new TokenGraphemeSplitter(ide).getTokenGraphemes( - input, - ); + test(`${input} -> ${displayOutput}`, async () => { + const splitter = new TokenGraphemeSplitter(ide, talonSpokenForms); + await splitter.ready; + const actualOutput = splitter.getTokenGraphemes(input); assert.deepEqual(actualOutput, expectedOutput); }); } }); } + +suite("Grapheme updates", () => { + test("only notifies when the normalized grapheme set changes", async () => { + let graphemes = ["ä", "ø"]; + // oxlint-disable-next-line unicorn/consistent-function-scoping + let update: () => void | Promise = () => { + // No-op + }; + const spokenForms: TalonSpokenForms = { + getSpokenForms: () => + FakeTalonSpokenForms.fromGraphemes(graphemes).getSpokenForms(), + onDidChange: (listener) => { + update = listener; + return { + dispose: () => { + // No-op + }, + }; + }, + }; + const ide = new FakeIDE(); + const splitter = new TokenGraphemeSplitter(ide, spokenForms); + await splitter.ready; + let notifications = 0; + splitter.registerAlgorithmChangeListener(() => { + notifications++; + }); + + try { + await update(); + assert.equal(notifications, 0); + + // Ordering, duplicates, default graphemes, and NFC normalization do not + // change the effective set. + graphemes = ["ø", "a\u0308", "ø", "a"]; + await update(); + assert.equal(notifications, 0); + + graphemes = ["ø"]; + await update(); + assert.equal(notifications, 1); + assert.equal(splitter.normalizeGrapheme("ä"), "a"); + + graphemes = ["ø", "ä"]; + await update(); + assert.equal(notifications, 2); + assert.equal(splitter.normalizeGrapheme("ä"), "ä"); + } finally { + ide.exit(); + } + }); +}); + +suite("Grapheme loading failures", () => { + test("initialization uses defaults when loading fails", async () => { + const spokenForms: TalonSpokenForms = { + getSpokenForms: () => Promise.reject(new SyntaxError("Invalid JSON")), + onDidChange: new Notifier().registerListener, + }; + const splitter = new TokenGraphemeSplitter(new FakeIDE(), spokenForms); + + await splitter.ready; + + assert.equal(splitter.normalizeGrapheme("A"), "a"); + assert.equal(splitter.normalizeGrapheme("ä"), "a"); + }); + + test("a failed update restores defaults and a later update recovers", async () => { + const notifier = new Notifier(); + let fail = false; + const customForms = FakeTalonSpokenForms.fromGraphemes(["ä"]); + const spokenForms: TalonSpokenForms = { + getSpokenForms: () => + fail + ? Promise.reject(new SyntaxError("Invalid JSON")) + : customForms.getSpokenForms(), + onDidChange: notifier.registerListener, + }; + const splitter = new TokenGraphemeSplitter(new FakeIDE(), spokenForms); + await splitter.ready; + assert.equal(splitter.normalizeGrapheme("ä"), "ä"); + + const update = () => + new Promise((resolve) => { + const disposable = splitter.registerAlgorithmChangeListener(() => { + disposable.dispose(); + resolve(); + }); + notifier.notifyListeners(); + }); + + fail = true; + await update(); + assert.equal(splitter.normalizeGrapheme("ä"), "a"); + + fail = false; + await update(); + assert.equal(splitter.normalizeGrapheme("ä"), "ä"); + }); +}); diff --git a/packages/lib-engine/src/tokenGraphemeSplitter/tokenGraphemeSplitter.ts b/packages/lib-engine/src/tokenGraphemeSplitter/tokenGraphemeSplitter.ts index 726e54bca1..4a9cb2782c 100644 --- a/packages/lib-engine/src/tokenGraphemeSplitter/tokenGraphemeSplitter.ts +++ b/packages/lib-engine/src/tokenGraphemeSplitter/tokenGraphemeSplitter.ts @@ -1,15 +1,16 @@ -import { deburr, escapeRegExp } from "lodash-es"; -import type { - Disposable, - IDE, - TokenHatSplittingMode, -} from "@cursorless/lib-common"; +import { deburr, isEqual } from "lodash-es"; +import type { Disposable, IDE, TalonSpokenForms } from "@cursorless/lib-common"; import { Notifier, matchAll } from "@cursorless/lib-common"; +import { asciiRange } from "../util/asciiRange"; /** - * A list of all symbols that are speakable by default in community. + * A list of all graphemes that are speakable by default in community. */ -const KNOWN_SYMBOLS = [ +const DEFAULT_GRAPHEMES = [ + // a–z + ...asciiRange(97, 122), + // 0–9 + ...asciiRange(48, 57), "!", "#", "$", @@ -44,21 +45,6 @@ const KNOWN_SYMBOLS = [ "£", '"', ]; -const KNOWN_SYMBOL_REGEXP_STR = KNOWN_SYMBOLS.map(escapeRegExp).join("|"); - -const KNOWN_GRAPHEME_REGEXP_STR = ["[a-zA-Z0-9]", KNOWN_SYMBOL_REGEXP_STR].join( - "|", -); - -/** - * Any token *not* matched by this regex will be mapped to {@link UNKNOWN}, so - * that they will count as the same grapheme from the perspective of hat - * allocation, and can be referred to using "special", "red special", etc. - */ -const KNOWN_GRAPHEME_MATCHER = new RegExp( - `^(${KNOWN_GRAPHEME_REGEXP_STR})$`, - "u", -); /** * All unknown graphemes will be mapped to this value, so that they will count @@ -75,41 +61,48 @@ export const GRAPHEME_SPLIT_REGEX = /\p{L}\p{M}*|[\p{N}\p{P}\p{S}]/gu; export class TokenGraphemeSplitter { private disposables: Disposable[] = []; private algorithmChangeNotifier = new Notifier(); - private tokenHatSplittingMode!: TokenHatSplittingMode; - - constructor(private ide: IDE) { + private graphemes: Set = new Set(DEFAULT_GRAPHEMES); + private updateGeneration = 0; + readonly ready: Promise; + + constructor( + private ide: IDE, + private talonSpokenForms: TalonSpokenForms, + ) { ide.disposeOnExit(this); - this.updateTokenHatSplittingMode = - this.updateTokenHatSplittingMode.bind(this); + this.updateGraphemes = this.updateGraphemes.bind(this); this.getTokenGraphemes = this.getTokenGraphemes.bind(this); + this.ready = this.updateGraphemes(); + this.disposables.push(talonSpokenForms.onDidChange(this.updateGraphemes)); + } - this.updateTokenHatSplittingMode(); + private async updateGraphemes() { + const generation = ++this.updateGeneration; + const customGraphemes = await this.getCustomGraphemes(); - this.disposables.push( - // Notify listeners in case the user changed their token hat splitting - // setting. - ide.configuration.onDidChangeConfiguration( - this.updateTokenHatSplittingMode, - ), - ); + if (generation !== this.updateGeneration) { + // Another update has occurred since this one started, so we should abort. + return; + } + + const graphemes = new Set([...DEFAULT_GRAPHEMES, ...customGraphemes]); + if (!isEqual(this.graphemes, graphemes)) { + this.graphemes = graphemes; + this.algorithmChangeNotifier.notifyListeners(); + } } - private updateTokenHatSplittingMode() { - const { lettersToPreserve, symbolsToPreserve, ...rest } = - this.ide.configuration.getOwnConfiguration("tokenHatSplittingMode"); - - this.tokenHatSplittingMode = { - lettersToPreserve: lettersToPreserve.map((grapheme) => - grapheme.toLowerCase().normalize("NFC"), - ), - symbolsToPreserve: symbolsToPreserve.map((grapheme) => - grapheme.normalize("NFC"), - ), - ...rest, - }; - - this.algorithmChangeNotifier.notifyListeners(); + private async getCustomGraphemes() { + try { + const spokenForms = await this.talonSpokenForms.getSpokenForms(); + return spokenForms.spokenForms + .filter((s) => s.type === "grapheme") + .map((s) => s.id.normalize("NFC")); + } catch { + // Any errors are already handled in `CustomSpokenForms`. + return []; + } } /** @@ -126,55 +119,49 @@ export class TokenGraphemeSplitter { })); /** - * Normalizes the grapheme {@link rawGraphemeText} based on user - * configuration. Proceeds as follows: + * Normalizes {@link rawGraphemeText} using the default graphemes and the + * graphemes exported by Talon. Proceeds as follows: * * 1. Runs text through Unicode NFC normalization to ensure that characters * that look identical are handled the same (eg whether they use combining * mark or single codepoint for diacritics). * 2. If the grapheme is a known grapheme, returns it. - * 3. Transforms grapheme to lowercase if - * {@link TokenHatSplittingMode.preserveCase} is `false` - * 3. Returns the (possibly case-normalised) grapheme if it appears in - * {@link TokenHatSplittingMode.lettersToPreserve} - * 4. Strips diacritics from the grapheme - * 5. If the grapheme doesn't match {@link KNOWN_GRAPHEME_MATCHER}, maps the - * grapheme to the constant {@link UNKNOWN}, so that it can be referred to - * using "special", "red special", etc. - * 6. Returns the grapheme. + * 3. Converts the grapheme to lowercase and returns it if it is known. + * 4. Strips diacritics and returns the resulting grapheme if it is known. + * 5. Returns {@link UNKNOWN} if none of these forms is known, so that the + * grapheme can be referred to using "special", "red special", etc. * * @param rawGraphemeText The raw grapheme text to normalise * @returns The normalised grapheme */ normalizeGrapheme(rawGraphemeText: string): string { - const { preserveCase, lettersToPreserve, symbolsToPreserve } = - this.tokenHatSplittingMode; - // We always normalise the grapheme so that the user doesn't get confusing // behaviour where the grapheme is represented as the naked grapheme and a // separate combining diacritic, but they pass in the combined version of // the grapheme. let returnValue = rawGraphemeText.normalize("NFC"); - if (symbolsToPreserve.includes(returnValue)) { + // 1) "Å" is a valid grapheme + if (this.graphemes.has(returnValue)) { return returnValue; } - if (!preserveCase) { - returnValue = returnValue.toLowerCase(); - } + returnValue = returnValue.toLowerCase(); - if (lettersToPreserve.includes(returnValue.toLowerCase())) { + // 2) "å" is a valid grapheme after converting to lowercase + if (this.graphemes.has(returnValue)) { return returnValue; } returnValue = deburr(returnValue); - if (!KNOWN_GRAPHEME_MATCHER.test(returnValue)) { - returnValue = UNKNOWN; + // 3) "a" is a valid grapheme after stripping diacritics + if (this.graphemes.has(returnValue)) { + return returnValue; } - return returnValue; + // 4) None of the above forms is known, so we return UNKNOWN + return UNKNOWN; } /** diff --git a/packages/lib-engine/src/util/asciiRange.ts b/packages/lib-engine/src/util/asciiRange.ts new file mode 100644 index 0000000000..111224e8b6 --- /dev/null +++ b/packages/lib-engine/src/util/asciiRange.ts @@ -0,0 +1,4 @@ +export const asciiRange = (start: number, end: number) => + Array.from({ length: end - start + 1 }, (_, i) => + String.fromCodePoint(start + i), + ); diff --git a/packages/lib-node-common/src/FileSystemTalonSpokenForms.ts b/packages/lib-node-common/src/FileSystemTalonSpokenForms.ts index f3ebce6fc8..288a6c1de5 100644 --- a/packages/lib-node-common/src/FileSystemTalonSpokenForms.ts +++ b/packages/lib-node-common/src/FileSystemTalonSpokenForms.ts @@ -15,11 +15,15 @@ const LATEST_SPOKEN_FORMS_JSON_VERSION = 1; export class FileSystemTalonSpokenForms implements TalonSpokenForms { private disposable: Disposable; private notifier = new Notifier(); + private payloadPromise?: Promise; constructor(private fileSystem: FileSystem) { this.disposable = this.fileSystem.watchDir( path.dirname(this.fileSystem.cursorlessTalonStateJsonPath), - () => this.notifier.notifyListeners(), + () => { + this.payloadPromise = undefined; + this.notifier.notifyListeners(); + }, ); } @@ -33,11 +37,33 @@ export class FileSystemTalonSpokenForms implements TalonSpokenForms { } async getSpokenForms(): Promise { + if (this.payloadPromise != null) { + return this.payloadPromise; + } + const promise = this.readStateFile(); + this.payloadPromise = promise; + try { + return await promise; + } catch (error) { + if (this.payloadPromise === promise) { + this.payloadPromise = undefined; + } + throw error; + } + } + + dispose() { + this.disposable.dispose(); + } + + private async readStateFile(): Promise { let payload: TalonSpokenFormsPayload; try { - payload = JSON.parse( - await readFile(this.fileSystem.cursorlessTalonStateJsonPath, "utf8"), + const stateFileContent = await readFile( + this.fileSystem.cursorlessTalonStateJsonPath, + "utf8", ); + payload = JSON.parse(stateFileContent); } catch (error) { if (isEnoentError(error)) { throw new NeedsInitialTalonUpdateError( @@ -56,8 +82,4 @@ export class FileSystemTalonSpokenForms implements TalonSpokenForms { return payload; } - - dispose() { - this.disposable.dispose(); - } } diff --git a/packages/lib-node-common/src/cheatsheet/Cheatsheet.ts b/packages/lib-node-common/src/cheatsheet/Cheatsheet.ts index 97ffe2e2b8..9b7a7adc28 100644 --- a/packages/lib-node-common/src/cheatsheet/Cheatsheet.ts +++ b/packages/lib-node-common/src/cheatsheet/Cheatsheet.ts @@ -1,13 +1,16 @@ import { readFile, writeFile } from "node:fs/promises"; import path from "node:path"; -import type { CheatsheetInfo, IDE } from "@cursorless/lib-common"; +import type { + CheatsheetInfo, + IDE, + TalonSpokenForms, +} from "@cursorless/lib-common"; import { getCheatsheetInfo, getDefaultCheatsheetInfo, getErrorMessage, showWarning, } from "@cursorless/lib-common"; -import type { FileSystemTalonSpokenForms } from "../FileSystemTalonSpokenForms"; import { injectCheatsheetInfo } from "./injectCheatsheetInfo"; interface CheatSheetCommandArgV0 { @@ -37,7 +40,7 @@ export type CheatSheetCommandArg = export async function showCheatsheet( ide: IDE, - talonSpokenForms: FileSystemTalonSpokenForms, + talonSpokenForms: TalonSpokenForms, arg: CheatSheetCommandArg, ) { const cheatsheetInfo = await getCheatsheetInfoForCommand( @@ -56,7 +59,7 @@ export async function showCheatsheet( async function getCheatsheetInfoForCommand( ide: IDE, - talonSpokenForms: FileSystemTalonSpokenForms, + talonSpokenForms: TalonSpokenForms, arg: CheatSheetCommandArg, ): Promise { const version = arg.version; diff --git a/packages/lib-node-common/src/runRecordedTest.ts b/packages/lib-node-common/src/runRecordedTest.ts index 04feb1f73c..954f3542a7 100644 --- a/packages/lib-node-common/src/runRecordedTest.ts +++ b/packages/lib-node-common/src/runRecordedTest.ts @@ -150,10 +150,13 @@ export async function runRecordedTest({ // Ensure that the expected hats are present // Ignore any allocation triggered by editor setup so that initial hats do not // depend on whether VS Code delivered those events before this point. - await hatTokenMap.allocateHats( - serializedMarksToTokenHats(fixture.initialState.marks, editor), - { startFresh: true }, - ); + await hatTokenMap.allocateHats({ + startFresh: true, + forceTokenHats: serializedMarksToTokenHats( + fixture.initialState.marks, + editor, + ), + }); await Promise.all( (fixture.initialState.highlights ?? []).map((highlight) => diff --git a/packages/lib-talonjs-core/src/ide/TalonJsConfiguration.ts b/packages/lib-talonjs-core/src/ide/TalonJsConfiguration.ts index 1b18389862..bc99803490 100644 --- a/packages/lib-talonjs-core/src/ide/TalonJsConfiguration.ts +++ b/packages/lib-talonjs-core/src/ide/TalonJsConfiguration.ts @@ -11,11 +11,6 @@ import type { import { HatStability } from "@cursorless/lib-common"; const CONFIGURATION_DEFAULTS: CursorlessConfiguration = { - tokenHatSplittingMode: { - preserveCase: false, - lettersToPreserve: [], - symbolsToPreserve: [], - }, wordSeparators: ["_"], decorationDebounceDelayMs: 50, experimental: { diff --git a/packages/lib-tutorial/src/setupStep.ts b/packages/lib-tutorial/src/setupStep.ts index f40b298e5d..d41ccf27b2 100644 --- a/packages/lib-tutorial/src/setupStep.ts +++ b/packages/lib-tutorial/src/setupStep.ts @@ -107,9 +107,9 @@ async function applySnapshot( ); // Ensure that the expected hats are present - await hatTokenMap.allocateHats( - serializedMarksToTokenHats(snapshot.marks, editor), - ); + await hatTokenMap.allocateHats({ + forceTokenHats: serializedMarksToTokenHats(snapshot.marks, editor), + }); await editableEditor.focus(); diff --git a/packages/test-vscode-e2e/src/suite/instanceAcrossSplit.vscode.test.ts b/packages/test-vscode-e2e/src/suite/instanceAcrossSplit.vscode.test.ts index 650dda9a56..0430cd99c3 100644 --- a/packages/test-vscode-e2e/src/suite/instanceAcrossSplit.vscode.test.ts +++ b/packages/test-vscode-e2e/src/suite/instanceAcrossSplit.vscode.test.ts @@ -107,19 +107,21 @@ async function runTest( const { document: fromDocument } = fromEditor; fromEditor.selections = [new Selection(0, 0, 0, 0)]; - await hatTokenMap.allocateHats([ - { - grapheme: "a", - hatStyle: "default", - hatRange: new Range(0, 0, 0, 1), - token: { - editor: instanceEditor, - offsets: { start: 0, end: 3 }, - range: new Range(0, 0, 0, 3), - text: "aaa", + await hatTokenMap.allocateHats({ + forceTokenHats: [ + { + grapheme: "a", + hatStyle: "default", + hatRange: new Range(0, 0, 0, 1), + token: { + editor: instanceEditor, + offsets: { start: 0, end: 3 }, + range: new Range(0, 0, 0, 3), + text: "aaa", + }, }, - }, - ]); + ], + }); // "from this" / "from file this", depending on the value of `useWholeFile` await runCursorlessCommand({ diff --git a/packages/tool-meta-updater/src/updateGraphemeDefaultSpokenFormsMd.ts b/packages/tool-meta-updater/src/updateGraphemeDefaultSpokenFormsMd.ts index 2e992343c3..18e8b6bdd6 100644 --- a/packages/tool-meta-updater/src/updateGraphemeDefaultSpokenFormsMd.ts +++ b/packages/tool-meta-updater/src/updateGraphemeDefaultSpokenFormsMd.ts @@ -38,6 +38,8 @@ export function updateGraphemeDefaultSpokenFormsMd( "", "Cursorless uses the [Talon Community](https://github.com/talonhub/community) alphabet, digits, and symbol spoken forms via the [`user.any_alphanumeric_key`](https://github.com/talonhub/community/blob/607c3415f5f29a5f75db6fe5648e37f514f62ac5/core/keys/keys.py#L71-L74) capture.", "", + "Additional characters provided by the lists in this capture are automatically preserved when allocating hats. See [Unicode support](unicode-support.md) for case handling, accented letters, and custom symbols.", + "", "## Alphabet", "", ...formatTable(HEADERS, alphabet),