Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 0 additions & 21 deletions packages/app-vscode/package.json
Original file line number Diff line number Diff line change
Expand Up @@ -334,27 +334,6 @@
"order": 2,
"description": "Keep a local, sanitized command history. This history is never sent to our servers, and any commands that may contain text will be sanitized. These statistics can be used in the future for doing local analyses to determine ways you can improve your Cursorless efficiency. We may also support a way for you to send your statistics to us for analysis in the future, but this will be opt-in only."
},
"cursorless.tokenHatSplittingMode.preserveCase": {
"type": "boolean",
"default": false,
"markdownDescription": "Whether to distinguish between uppercase and lower case letters for hats. Set this to `true` if you have separate terms for uppercase letters in your `<user.any_alphanumeric_key>` capture."
},
"cursorless.tokenHatSplittingMode.lettersToPreserve": {
"type": "array",
"items": {
"type": "string"
},
"default": [],
"description": "A list of characters whose accents should not be stripped. This can be used, for example, if you would like to strip all accents except for those of a few characters, which you would add to this list."
},
"cursorless.tokenHatSplittingMode.symbolsToPreserve": {
"type": "array",
"items": {
"type": "string"
},
"default": [],
"markdownDescription": "A list of symbols that shouldn't be normalized by the token hat splitter. Add any extra symbols here that you have added to your `<user.any_alphanumeric_key>` capture. Unlike the Accents To Preserve setting, these symbols won't even undergo case normalisation, so you would need separate terms for the lowercase and uppercase versions (if the symbol has a notion of upper and lower case)."
},
"cursorless.decorationDebounceDelayMs": {
"type": "number",
"default": 50,
Expand Down
18 changes: 18 additions & 0 deletions packages/app-vscode/src/extension.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@ import {
FakeCommandServerApi,
FakeIDE,
NormalizedIDE,
showWarning,
} from "@cursorless/lib-common";
import type { EngineProps } from "@cursorless/lib-engine";
import { CommandHistory, createCursorlessEngine } from "@cursorless/lib-engine";
Expand All @@ -26,6 +27,7 @@ import { createScopeVisualizer } from "./createScopeVisualizer";
import { createTreeSitter } from "./createTreeSitter";
import { createTutorial } from "./createTutorial";
import { createVscodeIde } from "./createVscodeIde";
import type { VscodeIDE } from "./ide/vscode/VscodeIDE";
import { InstallationDependencies } from "./InstallationDependencies";
import { KeyboardCommands } from "./keyboard/KeyboardCommands";
import { registerCommands } from "./registerCommands";
Expand Down Expand Up @@ -153,6 +155,8 @@ export async function activate(
hats,
);

deprecatedSettings(vscodeIDE);

registerCommands(
context,
vscodeIDE,
Expand Down Expand Up @@ -190,3 +194,17 @@ export async function activate(
: undefined,
};
}

function deprecatedSettings(ide: VscodeIDE) {
// DEPRECATED @ 2026-10-05
const value = vscodeApi.workspace
.getConfiguration("cursorless")
.get("tokenHatSplittingMode");
if (value != null) {
void showWarning(
ide.messages,
"tokenHatSplittingModeDeprecated",
"The 'cursorless.tokenHatSplittingMode' setting is deprecated and not needed when using an up to date Cursorless Talon",
);
}
}
8 changes: 3 additions & 5 deletions packages/app-vscode/src/registerCommands.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,14 +3,12 @@ import type {
CommandHistoryStorage,
CursorlessCommandId,
ScopeType,
TalonSpokenForms,
} from "@cursorless/lib-common";
import { CURSORLESS_COMMAND_ID } from "@cursorless/lib-common";
import type { CommandApi, StoredTargetMap } from "@cursorless/lib-engine";
import { analyzeCommandHistory } from "@cursorless/lib-engine";
import type {
CheatSheetCommandArg,
FileSystemTalonSpokenForms,
} from "@cursorless/lib-node-common";
import type { CheatSheetCommandArg } from "@cursorless/lib-node-common";
import { showCheatsheet } from "@cursorless/lib-node-common";
import type {
ScopeTestRecorder,
Expand Down Expand Up @@ -45,7 +43,7 @@ export function registerCommands(
tutorial: VscodeTutorial,
installationDependencies: InstallationDependencies,
storedTargets: StoredTargetMap,
talonSpokenForms: FileSystemTalonSpokenForms,
talonSpokenForms: TalonSpokenForms,
): void {
const runCommandWrapper = async (run: () => Promise<unknown>) => {
try {
Expand Down
2 changes: 2 additions & 0 deletions packages/app-web-docs/src/docs/user/alphabet-and-symbols.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,8 @@ sidebar_position: 7

Cursorless uses the [Talon Community](https://github.com/talonhub/community) alphabet, digits, and symbol spoken forms via the [`user.any_alphanumeric_key`](https://github.com/talonhub/community/blob/607c3415f5f29a5f75db6fe5648e37f514f62ac5/core/keys/keys.py#L71-L74) capture.

Additional characters provided by the lists in this capture are automatically preserved when allocating hats. See [Unicode support](unicode-support.md) for case handling, accented letters, and custom symbols.

## Alphabet

| Character | Default spoken form |
Expand Down
29 changes: 20 additions & 9 deletions packages/app-web-docs/src/docs/user/unicode-support.md
Original file line number Diff line number Diff line change
Expand Up @@ -5,31 +5,42 @@ sidebar_position: 2

# Unicode support

Cursorless has first-class support for Unicode. By default, when constructing hats, Cursorless will ignore capitalization and any accents or diacritics over letters. For example, each of the following four tokens could each be selected by saying `"take air"` if there were a gray hat over their first letter (note the accents on the first letter for some of them):
Cursorless has first-class support for Unicode. With the default Talon alphabet, Cursorless ignores capitalization and accents or diacritics when constructing hats. For example, each of the following four tokens could be selected by saying `"take air"` if there were a gray hat over its first letter:

- africa
- áfrica
- Africa
- África

For Unicode symbols that are not letters, and that are not speakable by default, for example emoji, Chinese characters, etc, we have a special "character" called `"special"` that can be used. So for example, if there were a blue hat over a '😄' character, you could say `"take blue special"` to select it. As always, the spoken form `"special"` can be [customized](customization.md).
Characters that do not have a known spoken form, even after normalization, can be referred to using `"special"`. For example, if there were a blue hat over a `😄` character, you could say `"take blue special"` to select it. This also works for letters that cannot be normalized to a known character, such as Chinese characters. As always, the spoken form `"special"` can be [customized](customization.md).

## Advanced customization

The above setup will allow you to refer to any Unicode token, and is sufficient for most users. However, if you have overridden your `<user.any_alphanumeric_key>` capture to contain characters other than lowercase letters and the default symbols, you can tell Cursorless to be less aggressive with its normalization, so that it can allocate hats more efficiently. Note that this is not necessary in order to refer to these tokens; it just makes hat allocation slightly more efficient.
With an up-to-date Cursorless Talon installation, Cursorless automatically preserves additional characters provided by the Talon lists used in your `<user.any_alphanumeric_key>` capture. Add your spoken forms there; no editor setting is needed. This lets Cursorless allocate hats separately for these characters instead of grouping them with normalized letters or `"special"`.

The old `cursorless.tokenHatSplittingMode` settings (`preserveCase`, `lettersToPreserve`, and `symbolsToPreserve`) are deprecated and can be removed from your editor settings.

### Preserving case

If you have a separate alphabet for uppercase letters as part of `<user.any_alphanumeric_key>`, you can enable the _Cursorless › Token Hat Splitting Mode: **Preserve Case**_ setting, and Cursorless will distinguish between lower and uppercase letters.
If your `<user.any_alphanumeric_key>` capture provides separate spoken forms for uppercase letters, Cursorless preserves those letters automatically. For example, if `"upper air"` produces `A`, a gray hat on the `A` in `Africa` can be addressed with `"take upper air"`, while a gray hat on lowercase `a` uses `"take air"`.

### Preserving special letters

If you have terms in `<user.any_alphanumeric_key>` for letters with accents, such as `é`, or other letters, such as `ø` or `ꝏ`, you can use the following setting:
If your capture provides terms for accented letters, such as `é`, or other letters, such as `ø` or `ꝏ`, Cursorless preserves them automatically.

For example, if `"a umlaut"` produces `ä` and you have no separate form for `Ä`, a gray hat over the first letter of either `ällo` or `Ällo` can be addressed with `"take a umlaut"`. If you also provide a spoken form for `Ä`, Cursorless treats it separately. Providing only `Ä` does not preserve lowercase `ä`; lowercase `ä` still normalizes to `a`.

### Preserving symbols

Symbols provided by your capture are also preserved automatically. For example, if `"sigma"` produces `σ` and `"upper sigma"` produces `Σ`, a blue hat on those characters can be addressed with `"take blue sigma"` and `"take blue upper sigma"`, respectively.

#### _Cursorless › Token Hat Splitting Mode: **Letters To Preserve**_
## Normalization order

Add any accented letters to this list that you have a spoken form for in `<user.any_alphanumeric_key>`. Cursorless will then preserve their accents during normalization. Note that Cursorless will still do case normalisation for these letters if you have [case preservation](#preserving-case) on. So, for example, if the list contains `ä`, and you'd like to refer to the token `Ällo` with a hat over the first letter (`Ä`), you can use your spoken form for `ä`.
Cursorless first normalizes each character to Unicode NFC so that equivalent representations, such as an accented letter written as one codepoint or with a combining mark, are treated the same. It then:

### _Cursorless › Token Hat Splitting Mode: **Symbols To Preserve**_
1. Preserves the character if it is a default character or is provided by Talon.
2. Otherwise, converts it to lowercase and uses that form if it is known.
3. Otherwise, strips accents and diacritics and uses the resulting form if it is known.
4. Otherwise, assigns it to `"special"`.

Any Unicode symbols in this list will not undergo any normalisation, even case normalisation. Use this list for symbols for which you have spoken forms in `<user.any_alphanumeric_key>` that shouldn't be normalised at all, even by case. For example, if you have spoken forms for `Σ` and `σ`, and would like Cursorless not to treat them the same, you can add them to this list.
If custom characters cannot be loaded from Talon, Cursorless uses the default alphabet, digits, and symbols.
34 changes: 34 additions & 0 deletions packages/lib-common/src/FakeTalonSpokenForms.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
import type {
SpokenFormEntry,
TalonSpokenForms,
TalonSpokenFormsPayload,
} from "@cursorless/lib-common";

export class FakeTalonSpokenForms implements TalonSpokenForms {
public static fromGraphemes(graphemes: string[]): FakeTalonSpokenForms {
return new FakeTalonSpokenForms(
graphemes.map((grapheme) => ({
type: "grapheme",
id: grapheme,
spokenForms: [],
})),
);
}

constructor(private spokenForms: SpokenFormEntry[]) {}

getSpokenForms(): Promise<TalonSpokenFormsPayload> {
return Promise.resolve({
version: -1,
spokenForms: this.spokenForms,
});
}

onDidChange() {
return {
dispose: () => {
// No-op
},
};
}
}
27 changes: 0 additions & 27 deletions packages/lib-common/src/ide/types/Configuration.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,6 @@ import type { Disposable } from "./ide.types";
import type { GetFieldType, Paths } from "./Paths";

export type CursorlessConfiguration = {
tokenHatSplittingMode: TokenHatSplittingMode;
wordSeparators: string[];
experimental: {
hatStability: HatStability;
Expand All @@ -19,11 +18,6 @@ export type CursorlessConfigKey = keyof CursorlessConfiguration;
export type ConfigurationScope = { languageId: string };

export const CONFIGURATION_DEFAULTS: CursorlessConfiguration = {
tokenHatSplittingMode: {
preserveCase: false,
lettersToPreserve: [],
symbolsToPreserve: [],
},
wordSeparators: ["_"],
decorationDebounceDelayMs: 50,
experimental: {
Expand Down Expand Up @@ -51,24 +45,3 @@ export interface Configuration {

onDidChangeConfiguration(listener: Listener): Disposable;
}

export interface TokenHatSplittingMode {
/**
* Whether to distinguished between uppercase and lower case letters for hat
*/
preserveCase: boolean;

/**
* A list of characters whose accents should not be stripped. This can be
* used, for example, if you would like to strip all accents except for those
* of a few characters, which you can add to this string.
*/
lettersToPreserve: string[];

/**
* A list of symbols that shouldn't be normalized by the token hat splitter.
* Add any extra symbols here that you have added to your
* <user.any_alphanumeric_key> capture.
*/
symbolsToPreserve: string[];
}
1 change: 1 addition & 0 deletions packages/lib-common/src/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@ export * from "./Debouncer";
export * from "./errors";
export * from "./extensionDependencies";
export * from "./FakeCommandServerApi";
export * from "./FakeTalonSpokenForms";
export * from "./ide/fake/FakeIDE";
export * from "./ide/inMemoryTextEditor/InMemoryTextDocument";
export * from "./ide/inMemoryTextEditor/InMemoryTextEditor";
Expand Down
10 changes: 6 additions & 4 deletions packages/lib-common/src/types/HatTokenMap.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,10 +7,7 @@ import type { Token } from "./Token";
* Maps from (hatStyle, character) pairs to tokens
*/
export interface HatTokenMap {
allocateHats(
forceTokenHats?: TokenHat[],
options?: HatAllocationOptions,
): Promise<void>;
allocateHats(options?: HatAllocationOptions): Promise<void>;
getReadableMap(usePrePhraseSnapshot: boolean): Promise<ReadOnlyHatMap>;
}

Expand All @@ -20,6 +17,11 @@ export interface HatAllocationOptions {
* Defaults to `false`. Forced hats still apply when starting fresh.
*/
startFresh?: boolean;

/** If supplied, force the allocator to use these hats
* for the given tokens. This is used for the tutorial, and for testing.
*/
forceTokenHats?: TokenHat[];
}

export interface TokenHat {
Expand Down
13 changes: 8 additions & 5 deletions packages/lib-engine/src/core/HatAllocator.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@ import {
Range,
tokenHatToPlainObject,
} from "@cursorless/lib-common";
import { DisabledTalonSpokenForms } from "../disabledComponents/DisabledTalonSpokenForms";
import { TokenGraphemeSplitter } from "../tokenGraphemeSplitter";
import { HatAllocator } from "./HatAllocator";
import { IndividualHatMap } from "./IndividualHatMap";
Expand All @@ -16,11 +17,12 @@ import { RangeUpdater } from "./updateSelections/RangeUpdater";
suite("HatAllocator", () => {
test("initial allocation is independent of earlier hat assignments", async () => {
const ide = new HatTestIDE();
const talonSpokenForms = new DisabledTalonSpokenForms();
ide.configuration.mockConfiguration("experimental", {
...ide.configuration.getOwnConfiguration("experimental"),
hatStability: HatStability.stable,
});
const splitter = new TokenGraphemeSplitter(ide);
const splitter = new TokenGraphemeSplitter(ide, talonSpokenForms);
const rangeUpdater = new RangeUpdater(ide);
const map = new IndividualHatMap(ide, splitter, rangeUpdater);
const hats: Hats = {
Expand Down Expand Up @@ -66,17 +68,18 @@ suite("HatAllocator", () => {

try {
// Establish the expected allocation with no earlier editor events.
await allocator.allocateHats([forcedWorld]);
await allocator.allocateHats({ forceTokenHats: [forcedWorld] });
const expected = snapshot();

// Simulate an allocation that occurred before fixture initialization.
await allocator.allocateHats([previousHello]);
await allocator.allocateHats([forcedWorld]);
await allocator.allocateHats({ forceTokenHats: [previousHello] });
await allocator.allocateHats({ forceTokenHats: [forcedWorld] });
assert.notDeepEqual(snapshot(), expected);
assert.ok(map.getToken("blue", "h"), "Normal allocation preserves hats");

await allocator.allocateHats([forcedWorld], {
await allocator.allocateHats({
startFresh: true,
forceTokenHats: [forcedWorld],
});
assert.deepEqual(snapshot(), expected);
assert.ok(map.getToken("default", "w"), "Forced hats are still applied");
Expand Down
28 changes: 15 additions & 13 deletions packages/lib-engine/src/core/HatAllocator.ts
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@ import type {
HatAllocationOptions,
Hats,
IDE,
TokenHat,
} from "@cursorless/lib-common";
import type { TokenGraphemeSplitter } from "../tokenGraphemeSplitter";
import { allocateHats } from "../util/allocateHats";
Expand All @@ -16,6 +15,7 @@ interface Context {

export class HatAllocator {
private disposables: Disposable[] = [];
private startFresh = false;

constructor(
private ide: IDE,
Expand All @@ -25,9 +25,10 @@ export class HatAllocator {
) {
ide.disposeOnExit(this);

const debouncer = new DecorationDebouncer(ide.configuration, () =>
this.allocateHats(),
);
const debouncer = new DecorationDebouncer(ide.configuration, () => {
this.allocateHats({ startFresh: this.startFresh });
this.startFresh = false;
});

this.disposables.push(
this.hats.onDidChangeEnabledHatStyles(debouncer.run),
Expand All @@ -47,9 +48,12 @@ export class HatAllocator {
ide.onDidChangeTextEditorSelection(debouncer.run),
// An Event which fires when the visible ranges of an editor has changed.
ide.onDidChangeTextEditorVisibleRanges(debouncer.run),
// Re-draw hats on grapheme splitting algorithm change in case they
// changed their token hat splitting setting.
tokenGraphemeSplitter.registerAlgorithmChangeListener(debouncer.run),
// Re-draw hats on grapheme splitting algorithm change.
tokenGraphemeSplitter.registerAlgorithmChangeListener(() => {
// When the grapheme splitting algorithm changes, we need to start fresh to ensure hats are correctly allocated.
this.startFresh = true;
debouncer.run();
}),

debouncer,
);
Expand All @@ -58,14 +62,12 @@ export class HatAllocator {
/**
* Allocate hats to the visible tokens.
*
* @param forceTokenHats If supplied, force the allocator to use these hats
* for the given tokens. This is used for the tutorial, and for testing.
* @param options Controls whether to start fresh without previous hat assignments.
*/
async allocateHats(
forceTokenHats?: TokenHat[],
{ startFresh = false }: HatAllocationOptions = {},
) {
async allocateHats({
startFresh = false,
forceTokenHats,
}: HatAllocationOptions = {}) {
const activeMap = await this.context.getActiveMap();

// Forced graphemes won't have been normalized
Expand Down
Loading
Loading