diff --git a/Cargo.lock b/Cargo.lock index 637da536..4d67b484 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3593,8 +3593,10 @@ version = "25.7.1" dependencies = [ "helix-core", "helix-loader", + "helix-stdx", "helix-term", "helix-view", + "ropey", "toml", ] diff --git a/book/src/SUMMARY.md b/book/src/SUMMARY.md index 9d060c2a..ec3a9df7 100644 --- a/book/src/SUMMARY.md +++ b/book/src/SUMMARY.md @@ -15,6 +15,7 @@ - [Keymap](./keymap.md) - [Command line](./command-line.md) - [Commands](./commands.md) + - [Language servers](./lsp.md) - [Language support](./lang-support.md) - [Workspace trust](./workspace-trust.md) - [Ecosystem](./ecosystem.md) @@ -27,6 +28,7 @@ - [Languages](./languages.md) - [Guides](./guides/README.md) - [Adding languages](./guides/adding_languages.md) + - [Adding highlight queries](./guides/highlights.md) - [Adding locals queries](./guides/locals.md) - [Adding textobject queries](./guides/textobject.md) - [Adding indent queries](./guides/indent.md) diff --git a/book/src/from-vim.md b/book/src/from-vim.md index f0256d20..df8ca6d1 100644 --- a/book/src/from-vim.md +++ b/book/src/from-vim.md @@ -1,10 +1,10 @@ # Migrating from Vim -- [Delete/Change Commands](#Delete/Change%20Commands) -- [Navigation](#Navigation) -- [Line Deletes](#Line%20Deletes) -- [Comment lines, Completion, Search](#Comment%20lines,%20Completion,%20Search) -- [File actions](#File%20actions) +- [Delete/Change Commands](#deletechange-commands) +- [Navigation](#navigation) +- [Line Deletes](#line-deletes) +- [Comment lines, Completion, Search](#comment-lines-completion-search) +- [File actions](#file-actions) Helix's editing model is strongly inspired from Vim and Kakoune, and a notable difference from Vim (and the most striking similarity to Kakoune) is that Helix diff --git a/book/src/guides/README.md b/book/src/guides/README.md index e53983d6..4e076a62 100644 --- a/book/src/guides/README.md +++ b/book/src/guides/README.md @@ -1,4 +1,5 @@ # Guides -This section contains guides for adding new language server configurations, -tree-sitter grammars, textobject and rainbow bracket queries, and other similar items. +This section contains guides for adding new languages to Helix: language and +grammar configuration, and the tree-sitter query files that drive highlighting, +power indentation, textobjects, symbol tags, and other features. diff --git a/book/src/guides/adding_languages.md b/book/src/guides/adding_languages.md index f9824215..6eca0283 100644 --- a/book/src/guides/adding_languages.md +++ b/book/src/guides/adding_languages.md @@ -36,7 +36,26 @@ below. 3. Refer to the [tree-sitter website](https://tree-sitter.github.io/tree-sitter/3-syntax-highlighting.html#highlights) for more information on writing queries. -4. A list of highlight captures can be found [on the themes page](https://docs.helix-editor.com/themes.html#scopes). +4. The highlight captures (`@function`, `@type`, ...) and how they resolve are + documented [on the themes page](./themes.md): + match the most specific scope that fits, capture the leaf node you mean, and + remember that the last matching pattern (and the innermost node) wins. +5. Helix loads several query files from that directory; only `highlights.scm` is + required: + + | File | Purpose | Guide | + |---|---|---| + | `highlights.scm` | syntax highlighting | [highlights.md](./highlights.md) | + | `injections.scm` | embed other languages in regions (strings, code fences) | [injection.md](./injection.md) | + | `indents.scm` | indentation | [indent.md](./indent.md) | + | `textobjects.scm` | textobjects and navigation (`mif`, `]f`, …) | [textobject.md](./textobject.md) | + | `locals.scm` | scope tracking so locals highlight distinctly | [locals.md](./locals.md) | + | `tags.scm` | document/workspace symbol pickers | [tags.md](./tags.md) | + | `rainbows.scm` | rainbow brackets | [rainbow_bracket_queries.md](./rainbow_bracket_queries.md) | + + A query file may reuse another language's with `; inherits: ` on the + first line. Run `cargo xtask query-check [language]` to check that the queries + are valid against the grammar. ## Common issues @@ -47,3 +66,6 @@ below. - If a parser is causing a segfault, or you want to remove it, make sure to remove the compiled parser located at `runtime/grammars/.so`. - If you are attempting to add queries and Helix is unable to locate them, ensure that the environment variable `HELIX_RUNTIME` is set to the location of the `runtime` folder you're developing in. +- Validate queries with `cargo xtask query-check [language]` (every query file + must compile against the grammar). `highlight-check` and `indent-check` + additionally run the real highlighter and indenter over the test fixtures catch mistakes. diff --git a/book/src/guides/highlights.md b/book/src/guides/highlights.md new file mode 100644 index 00000000..bbac4764 --- /dev/null +++ b/book/src/guides/highlights.md @@ -0,0 +1,53 @@ +## Adding highlight queries + +`highlights.scm` queries assign a highlight scope (`@function`, `@type`, +`@keyword`, ...) to nodes in the syntax tree; the theme then maps each scope to a +colour. Highlighting is the one query file that every language needs. + +Query files should be placed in `runtime/queries/{language}/highlights.scm` when +contributing to Helix. + +## Scopes + +The full list of highlight scopes, and what each is for, is documented on the +[themes page]. Match the most specific scope that fits the node — for example a +method call is `@function.method` while a plain field access is +`@variable.other.member`. + +A query file may reuse another language's with `; inherits: ` on the first +line (for example `tsx` inherits `typescript`, which inherits `ecma`). An +inherited file is compiled against *each* inheriting grammar, so every capture +must be valid there too. + +## Precedence + +Two rules decide which capture wins when more than one matches the same text: + +1. **Same span: last match wins.** Among captures covering the same bytes, the + pattern that appears later in the file wins. Put a generic rule *before* the + specific one that should override it. +2. **Nested nodes: innermost wins.** When a parent and a child node both cover + the text, the child's capture wins, regardless of file order. + +A common consequence of rule 2: capture the *leaf* you mean. A call captured on +a wrapping node loses to a base `(identifier) @variable` on the inner identifier, +so put `@function` on the identifier itself. + +Where the grammar can't distinguish a scope, casing is often used as a heuristic: + +```scm +((identifier) @constant + (#match? @constant "^[A-Z][A-Z_]*$")) +``` + +## Testing + +`cargo xtask query-check [language]` confirms the queries are valid against the +grammar. `cargo xtask highlight-check [language]` runs the real highlighter over +the fixtures in `tests/query/highlights//.`, where a +caret comment line (`// ^ @capture`) asserts the winning scope at the column +above it; this catches the precedence mistakes that `query-check` cannot see. +`cargo xtask highlight-check --dump ` prints the winning capture +per span for an arbitrary file. + +[themes page]: https://docs.helix-editor.com/themes.html#scopes diff --git a/book/src/guides/indent.md b/book/src/guides/indent.md index 230b94f7..769b9bc8 100644 --- a/book/src/guides/indent.md +++ b/book/src/guides/indent.md @@ -99,6 +99,10 @@ level for the line. captured, only the extension of the innermost one is prevented. All other ancestors are unaffected (regardless of whether the innermost ancestor would actually have been extended). +- `@opaque`: + Mark a literal body such as a string, heredoc, or block comment. Lines that + begin inside the captured node keep their existing indentation instead of being + reindented, so the contents of multi-line literals are left untouched. #### `@indent` / `@outdent` @@ -251,6 +255,32 @@ To help, we need to signal an end to the extension. We can do this with (return_statement) @extend.prevent-once ``` +#### Brace-less bodies + +A brace-less single-statement body: `if (cond)` with its statement on the next +line and no `{}` is a *following sibling* of the header, so the upward +traversal from the line above never reaches it. Capture the body directly and +give it the `all` scope: + +```scm +(if_statement + consequence: (_) @indent + (#not-kind-eq? @indent "compound_statement") + (#set! "scope" "all")) +(while_statement + body: (_) @indent + (#not-kind-eq? @indent "compound_statement") + (#set! "scope" "all")) +``` + +When a new line is typed right after the header, Helix descends into the +field-named body (`body`, `consequence`, or `alternative`) that the line opens, +so the body's own `@indent` governs it — no wrapper node or `@extend` is needed. +The `#not-kind-eq?` guard skips the braced form, which the surrounding block +already indents. Give `else` and `do .. while` their own pattern (on the +`alternative` / `body` field) so the trailing `else` / `while` keyword line is +not indented along with the body. + #### `@indent.always` / `@outdent.always` As mentioned before, normally if there is more than one `@indent` or `@outdent` @@ -372,3 +402,10 @@ Then, on the closing brace, we encounter an outdent with a scope of "all", which means the first line is included, and the indent level is cancelled out on this line. (Note these scopes are the defaults for `@indent` and `@outdent`—they are written explicitly for demonstration.) + +## Testing + +`cargo xtask indent-check [language]` checks the queries against the fixtures in +`tests/indent/.` in both modes: re-indenting each line +and simulating a newline typed after it, so a rule that is correct one way but +wrong the other is caught. diff --git a/book/src/guides/injection.md b/book/src/guides/injection.md index c94dadba..cf2e8921 100644 --- a/book/src/guides/injection.md +++ b/book/src/guides/injection.md @@ -4,6 +4,11 @@ Writing language injection queries allows one to highlight a specific node as a In addition to the [standard][upstream-docs] language injection options used by tree-sitter, there are a few Helix specific extensions that allow for more control. +Injection drives more than highlighting: within an injected region Helix also +uses the injected language's own indentation, textobjects, and comment tokens — +so, for example, editing JavaScript inside an HTML ` + + diff --git a/tests/query/highlights/vue/template.vue b/tests/query/highlights/vue/template.vue new file mode 100644 index 00000000..cd6fe15a --- /dev/null +++ b/tests/query/highlights/vue/template.vue @@ -0,0 +1,14 @@ + diff --git a/xtask/Cargo.toml b/xtask/Cargo.toml index 5421aca5..e0c5bd96 100644 --- a/xtask/Cargo.toml +++ b/xtask/Cargo.toml @@ -16,4 +16,6 @@ helix-term = { path = "../helix-term" } helix-core = { path = "../helix-core" } helix-view = { path = "../helix-view" } helix-loader = { path = "../helix-loader" } +helix-stdx = { path = "../helix-stdx" } +ropey.workspace = true toml.workspace = true diff --git a/xtask/src/main.rs b/xtask/src/main.rs index 7bb7e8c9..5a7e9919 100644 --- a/xtask/src/main.rs +++ b/xtask/src/main.rs @@ -45,6 +45,201 @@ pub mod tasks { Ok(()) } + pub fn indentcheck(languages: impl Iterator) -> Result<(), DynError> { + use helix_core::{ + indent::{ + is_opaque_interior, is_outdent_token_at, treesitter_indent_for_pos, IndentStyle, + }, + Syntax, + }; + use helix_stdx::rope::RopeSliceExt; + use ropey::Rope; + + let filter: HashSet = languages.collect(); + let loader = helix_core::config::default_lang_loader(); + let corpus = crate::path::tests_indent(); + let tab_width = 4; + let mut errors = 0usize; + let mut over_notes = 0usize; + + let mut entries: Vec<_> = std::fs::read_dir(&corpus)? + .filter_map(Result::ok) + .map(|e| e.path()) + .filter(|p| p.is_file()) + .collect(); + entries.sort(); + + for path in entries { + let stem = path.file_stem().and_then(|s| s.to_str()).unwrap_or(""); + if !filter.is_empty() && !filter.contains(stem) { + continue; + } + let file = path.file_name().and_then(|s| s.to_str()).unwrap_or(""); + + // Corpus files are named .; resolve the language by id. + let language = match loader + .languages() + .find(|(_, data)| data.config().language_id == stem) + { + Some((language, _)) => language, + None => { + return Err(format!( + "{file}: no configured language with id '{stem}' (corpus files are named .)" + ) + .into()) + } + }; + + let config = loader.language(language).config(); + let indent_style = IndentStyle::from_str( + &config + .indent + .as_ref() + .ok_or_else(|| format!("{file}: language '{stem}' has no indent config"))? + .unit, + ); + // Lines that are commented out are skipped: they self-document edge + // cases (e.g. known indent limitations) without being asserted on. + let comment_tokens: Vec = config.comment_tokens.clone().unwrap_or_default(); + + let doc = Rope::from_reader(&mut std::fs::File::open(&path)?)?; + let text = doc.slice(..); + let syntax = Syntax::new(text, language, &loader) + .map_err(|e| format!("{file}: failed to parse: {e:?}"))?; + let indent_query = loader + .indent_query(language) + .ok_or_else(|| format!("{file}: language '{stem}' has no indent query"))?; + + for i in 0..doc.len_lines() { + let line = text.line(i); + let Some(pos) = line.first_non_whitespace_char() else { + continue; + }; + + let trimmed = line.slice(pos..).to_string(); + if comment_tokens + .iter() + .any(|tok| trimmed.starts_with(tok.as_str())) + { + continue; + } + + let suggested = treesitter_indent_for_pos( + indent_query, + &syntax, + tab_width, + indent_style.indent_width(tab_width), + text, + i, + text.line_to_char(i) + pos, + false, + ) + .unwrap() + .to_string(&indent_style, tab_width); + + let actual = line + .get_slice(..pos) + .map(|s| s.to_string()) + .unwrap_or_default(); + + if actual != suggested { + errors += 1; + println!( + "{file}:{}: reindent expected {} columns, computed {}", + i + 1, + actual.chars().count(), + suggested.chars().count(), + ); + } + + // Typing direction: simulate pressing Enter at the end of this line and check the indent computed for the next line. + // - under-indent (computed < canonical) is always a failure: nothing pulls the line further in, so the user is left + // under-indented. + // - over-indent (computed > canonical) is only acceptable when the next line's leading token is @outdent (a closing + // bracket, case/else/except keyword, ...): entering it dedents the line. An over-indent on a plain statement (e.g. + // a line after `return` that should leave the block but doesn't) has nothing to correct it and is a real failure. + if i + 1 < doc.len_lines() { + if let Some(next_pos) = text.line(i + 1).first_non_whitespace_char() { + let next = text.line(i + 1); + let next_trim = next.slice(next_pos..).to_string(); + // Lines inside an @opaque body (string/comment) carry literal leading whitespace, not code indent — don't + // assert a typing indent for them. + let next_byte = + text.char_to_byte(text.line_to_char(i + 1) + next_pos) as u32; + + let next_is_opaque = + is_opaque_interior(indent_query, &syntax, text, next_byte); + let next_is_comment = next_is_opaque + || comment_tokens + .iter() + .any(|tok| next_trim.starts_with(tok.as_str())); + + if !next_is_comment { + let typed = treesitter_indent_for_pos( + indent_query, + &syntax, + tab_width, + indent_style.indent_width(tab_width), + text, + i, + text.line_to_char(i + 1) - 1, + true, + ) + .unwrap() + .to_string(&indent_style, tab_width); + + let next_actual = next + .get_slice(..next_pos) + .map(|s| s.to_string()) + .unwrap_or_default(); + + let typed_cols = typed.chars().count(); + let want_cols = next_actual.chars().count(); + let leading_outdent = || { + let byte = text.char_to_byte(text.line_to_char(i + 1) + next_pos); + is_outdent_token_at(indent_query, &syntax, text, byte as u32) + }; + + if typed_cols < want_cols { + errors += 1; + println!( + "{file}:{}: typing under-indents: computed {} columns, expected {} | {}", + i + 2, + typed_cols, + want_cols, + next_trim.trim_end(), + ); + } else if typed_cols > want_cols && !leading_outdent() { + // Over-indents are reported but not failed: every over-indent sits at a legitimate dedent point, and + // for indent-delimited languages (python after a block) the editor genuinely cannot know how far to + // dedent. Surfaced so a regression shows up in the diff; gate on under-indents only. + over_notes += 1; + println!( + "{file}:{}: note: typing over-indents: computed {} columns, expected {} (no leading outdent; review) | {}", + i + 2, + typed_cols, + want_cols, + next_trim.trim_end(), + ); + } + } + } + } + } + } + + if over_notes > 0 { + println!("Indent check: {over_notes} typing over-indent note(s) (not failures; review for regressions)"); + } + match errors { + 0 => { + println!("Indent check succeeded"); + Ok(()) + } + n => Err(format!("Indent check failed: {n} line(s) with wrong indentation").into()), + } + } + pub fn themecheck(themes: impl Iterator) -> Result<(), DynError> { use helix_view::theme::Loader; @@ -92,6 +287,9 @@ Usage: Run with `cargo xtask `, eg. `cargo xtask docgen`. docgen Generate files to be included in the mdbook output. query-check [languages] Check that tree-sitter queries are valid for the given languages, or all languages if none are specified. + indent-check [languages] Check indentation for the corpus files in tests/indent/ + (named .) against the configured grammars, + for the given languages, or all corpus files if none are specified. theme-check [themes] Check that the theme files in runtime/themes/ are valid for the given themes, or all themes if none are specified. " @@ -107,6 +305,7 @@ fn main() -> Result<(), DynError> { Some(t) => match t.as_str() { "docgen" => tasks::docgen()?, "query-check" => tasks::querycheck(args)?, + "indent-check" => tasks::indentcheck(args)?, "theme-check" => tasks::themecheck(args)?, invalid => return Err(format!("Invalid task name: {}", invalid).into()), }, diff --git a/xtask/src/path.rs b/xtask/src/path.rs index a46bdda0..a26a343c 100644 --- a/xtask/src/path.rs +++ b/xtask/src/path.rs @@ -22,3 +22,7 @@ pub fn ts_queries() -> PathBuf { pub fn themes() -> PathBuf { runtime().join("themes") } + +pub fn tests_indent() -> PathBuf { + project_root().join("tests").join("indent") +}