diff --git a/.goreleaser.yaml b/.goreleaser.yaml index aa5295fb..210d01a1 100644 --- a/.goreleaser.yaml +++ b/.goreleaser.yaml @@ -295,12 +295,6 @@ after: - cmd: cargo publish -p sugarloaf if: "{{ and .IsRelease .IsMerging }}" output: true - - cmd: cargo publish -p rio-proc-macros - if: "{{ and .IsRelease .IsMerging }}" - output: true - - cmd: cargo publish -p copa - if: "{{ and .IsRelease .IsMerging }}" - output: true - cmd: cargo publish -p corcovado if: "{{ and .IsRelease .IsMerging }}" output: true diff --git a/Cargo.lock b/Cargo.lock index 1fdc92f9..5e94c755 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -851,18 +851,6 @@ version = "0.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3d52eff69cd5e647efe296129160853a42795992097e8af39800e1060caeea9b" -[[package]] -name = "copa" -version = "0.4.2" -dependencies = [ - "arrayvec", - "criterion", - "memchr", - "rio-proc-macros", - "serde", - "simdutf8", -] - [[package]] name = "copypasta" version = "0.10.2" @@ -4149,7 +4137,6 @@ dependencies = [ "base64", "bitflags 2.11.1", "bytemuck", - "copa", "copypasta", "corcovado", "cursor-icon", @@ -4201,14 +4188,6 @@ dependencies = [ "zbus", ] -[[package]] -name = "rio-proc-macros" -version = "0.4.2" -dependencies = [ - "proc-macro2", - "quote", -] - [[package]] name = "rio-window" version = "0.4.2" @@ -4270,7 +4249,6 @@ dependencies = [ "ahash", "bitflags 2.11.1", "clap", - "copa", "corcovado", "cpal", "criterion", diff --git a/Cargo.toml b/Cargo.toml index 31fc6865..5283651d 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -3,8 +3,6 @@ members = [ "sugarloaf", "teletypewriter", "corcovado", - "copa", - "rio-proc-macros", "rio-backend", "rio-grapheme-width", "rio-window", @@ -38,8 +36,6 @@ rio-grapheme-width = { path = "rio-grapheme-width", version = "0.4.2" } sugarloaf = { path = "sugarloaf", version = "0.4.2" } # Own dependencies -copa = { path = "copa", default-features = true, version = "0.4.2" } -rio-proc-macros = { path = "rio-proc-macros", version = "0.4.2" } corcovado = { path = "corcovado", version = "0.4.2" } raw-window-handle = { version = "0.6.2", features = ["std"] } parking_lot = { version = "0.12.5", features = [ diff --git a/copa/Cargo.toml b/copa/Cargo.toml deleted file mode 100644 index bf2bdb21..00000000 --- a/copa/Cargo.toml +++ /dev/null @@ -1,37 +0,0 @@ -[package] -authors = ["Raphael Amorim "] -description = "Parser for implementing terminal emulators" -repository = "https://github.com/raphamorim/rio" -documentation = "https://github.com/raphamorim/rio" -readme = "README.md" -license = "Apache-2.0 OR MIT" -version = { workspace = true } -name = "copa" -edition = "2021" - -[features] -default = ["std"] -std = ["memchr/std"] -serde = ["dep:serde"] - -[dependencies] -arrayvec = { version = "0.7.6", default-features = false } -memchr = { version = "2.8.0", default-features = false } -serde = { workspace = true, optional = true } -simdutf8 = { version = "0.1.5", default-features = false } -rio-proc-macros = { workspace = true } - -[dev-dependencies] -criterion = { workspace = true } - -[[bench]] -name = "parser_benchmark" -harness = false - -[[example]] -name = "benchmark_comparison" -required-features = [] - -[[example]] -name = "advanced_benchmark" -required-features = [] diff --git a/copa/ORIGINAL-LICENSE-APACHE b/copa/ORIGINAL-LICENSE-APACHE deleted file mode 100644 index 1b5ec8b7..00000000 --- a/copa/ORIGINAL-LICENSE-APACHE +++ /dev/null @@ -1,176 +0,0 @@ - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - -TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - -1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - -2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - -3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - -4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - -5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - -6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - -7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - -8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - -9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - -END OF TERMS AND CONDITIONS diff --git a/copa/ORIGINAL-LICENSE-MIT b/copa/ORIGINAL-LICENSE-MIT deleted file mode 100644 index bb419c21..00000000 --- a/copa/ORIGINAL-LICENSE-MIT +++ /dev/null @@ -1,25 +0,0 @@ -Copyright (c) 2016 Joe Wilm - -Permission is hereby granted, free of charge, to any -person obtaining a copy of this software and associated -documentation files (the "Software"), to deal in the -Software without restriction, including without -limitation the rights to use, copy, modify, merge, -publish, distribute, sublicense, and/or sell copies of -the Software, and to permit persons to whom the Software -is furnished to do so, subject to the following -conditions: - -The above copyright notice and this permission notice -shall be included in all copies or substantial portions -of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF -ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED -TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A -PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT -SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY -CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION -OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR -IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER -DEALINGS IN THE SOFTWARE. diff --git a/copa/README.md b/copa/README.md deleted file mode 100644 index 5e44c3ce..00000000 --- a/copa/README.md +++ /dev/null @@ -1,13 +0,0 @@ -# Copa - -Copa is a fork of [Alacritty's VTE](https://github.com/alacritty/vte/) intended to extend [Paul Williams' ANSI parser state -machine] with custom instructions. - -The state machine doesn't assign meaning to the parsed data and is -thus not itself sufficient for writing a terminal emulator. Instead, it is -expected that an implementation of the `Perform` trait which does something useful with the parsed data. The `Parser` handles the book keeping, and the `Perform` gets to simply handle actions. - -See the [ansicode.txt](resources/ansicode.txt) for more info. - -[Paul Williams' ANSI parser state machine]: https://vt100.net/emu/dec_ansi_parser -[docs]: https://docs.rs/crate/vte/ diff --git a/copa/benches/parser_benchmark.rs b/copa/benches/parser_benchmark.rs deleted file mode 100644 index 2eb66348..00000000 --- a/copa/benches/parser_benchmark.rs +++ /dev/null @@ -1,384 +0,0 @@ -use copa::{Params, Parser, Perform}; -use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion}; -use std::hint::black_box as std_black_box; - -/// A minimal performer that does nothing to avoid overhead in benchmarks -struct NoOpPerformer; - -impl Perform for NoOpPerformer { - fn print(&mut self, _c: char) {} - fn execute(&mut self, _byte: u8) {} - fn hook( - &mut self, - _params: &Params, - _intermediates: &[u8], - _ignore: bool, - _action: char, - ) { - } - fn put(&mut self, _byte: u8) {} - fn unhook(&mut self) {} - fn osc_dispatch(&mut self, _params: &[&[u8]], _bell_terminated: bool) {} - fn csi_dispatch( - &mut self, - _params: &Params, - _intermediates: &[u8], - _ignore: bool, - _action: char, - ) { - } - fn esc_dispatch(&mut self, _intermediates: &[u8], _ignore: bool, _byte: u8) {} -} - -/// Generate test data with various UTF-8 scenarios -fn generate_test_data() -> Vec<(&'static str, Vec)> { - vec![ - // ASCII only - ( - "ascii_text", - b"Hello, World! This is a simple ASCII text.".to_vec(), - ), - // Mixed ASCII and UTF-8 - ( - "mixed_utf8", - "Hello 🌍! This is mixed ASCII and UTF-8: café, naïve, 中文" - .as_bytes() - .to_vec(), - ), - // Heavy UTF-8 content - ( - "heavy_utf8", - "🎉🦀🚀 Rust is amazing! 中文测试 العربية русский язык 🌟✨💫" - .as_bytes() - .to_vec(), - ), - // Terminal escape sequences with UTF-8 - ( - "escape_sequences", - b"\x1b[31mRed text\x1b[0m Normal \x1b[32m\xF0\x9F\x8C\xB1 Green\x1b[0m" - .to_vec(), - ), - // OSC sequences with UTF-8 - ( - "osc_utf8", - b"\x1b]2;Terminal Title with UTF-8: \xF0\x9F\x92\xBB\x07".to_vec(), - ), - // CSI sequences - ( - "csi_sequences", - b"\x1b[1;32mBold Green\x1b[0m \x1b[4mUnderlined\x1b[0m".to_vec(), - ), - // Large text block (simulating real terminal output) - ("large_text", { - let mut data = Vec::new(); - for i in 0..1000 { - data.extend_from_slice( - format!("Line {}: Hello 🌍 World! 中文 {}\n", i, "🦀".repeat(5)) - .as_bytes(), - ); - } - data - }), - // Vim-like output (complex escape sequences) - ("vim_like", { - let mut data = Vec::new(); - // Simulate vim startup with lots of escape sequences - data.extend_from_slice( - b"\x1b[?1049h\x1b[22;0;0t\x1b[1;24r\x1b[?12h\x1b[?12l", - ); - data.extend_from_slice( - b"\x1b[22;2t\x1b[22;1t\x1b[27m\x1b[23m\x1b[29m\x1b[m\x1b[H\x1b[2J", - ); - data.extend_from_slice("VIM - Vi IMproved 🚀 version 9.0".as_bytes()); - data.extend_from_slice(b"\x1b[1;1H\x1b[42m\x1b[30m NORMAL \x1b[m"); - data - }), - // Partial UTF-8 sequences (stress test) - ("partial_utf8", { - let mut data = Vec::new(); - // Add some valid UTF-8 - data.extend_from_slice("Valid: 🦀".as_bytes()); - // Add partial UTF-8 that would be completed in next chunk - data.extend_from_slice(&[0xF0, 0x9F]); // Partial 4-byte UTF-8 - data - }), - // Invalid UTF-8 mixed with valid - ("invalid_utf8", { - let mut data = Vec::new(); - data.extend_from_slice(b"Valid text "); - data.extend_from_slice(&[0xFF, 0xFE]); // Invalid UTF-8 - data.extend_from_slice(" more valid text".as_bytes()); - data - }), - ] -} - -fn bench_parser_advance(c: &mut Criterion) { - let test_data = generate_test_data(); - - let mut group = c.benchmark_group("parser_advance"); - - for (name, data) in test_data.iter() { - group.bench_with_input(BenchmarkId::new("advance", name), data, |b, data| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, std_black_box(data)); - }); - }); - } - - group.finish(); -} - -fn bench_parser_advance_chunked(c: &mut Criterion) { - let test_data = generate_test_data(); - - let mut group = c.benchmark_group("parser_advance_chunked"); - - for (name, data) in test_data.iter() { - if data.len() < 100 { - continue; - } // Skip small data for chunked tests - - group.bench_with_input(BenchmarkId::new("chunked_8", name), data, |b, data| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - - // Process in 8-byte chunks to stress UTF-8 handling - for chunk in data.chunks(8) { - parser.advance(&mut performer, std_black_box(chunk)); - } - }); - }); - - group.bench_with_input(BenchmarkId::new("chunked_64", name), data, |b, data| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - - // Process in 64-byte chunks - for chunk in data.chunks(64) { - parser.advance(&mut performer, std_black_box(chunk)); - } - }); - }); - } - - group.finish(); -} - -fn bench_parser_advance_until_terminated(c: &mut Criterion) { - struct TerminatingPerformer { - count: usize, - terminate_at: usize, - } - - impl TerminatingPerformer { - fn new(terminate_at: usize) -> Self { - Self { - count: 0, - terminate_at, - } - } - } - - impl Perform for TerminatingPerformer { - fn print(&mut self, _c: char) { - self.count += 1; - } - - fn execute(&mut self, _byte: u8) { - self.count += 1; - } - - fn terminated(&self) -> bool { - self.count >= self.terminate_at - } - - fn hook( - &mut self, - _params: &Params, - _intermediates: &[u8], - _ignore: bool, - _action: char, - ) { - } - fn put(&mut self, _byte: u8) {} - fn unhook(&mut self) {} - fn osc_dispatch(&mut self, _params: &[&[u8]], _bell_terminated: bool) {} - fn csi_dispatch( - &mut self, - _params: &Params, - _intermediates: &[u8], - _ignore: bool, - _action: char, - ) { - } - fn esc_dispatch(&mut self, _intermediates: &[u8], _ignore: bool, _byte: u8) {} - } - - let test_data = generate_test_data(); - - let mut group = c.benchmark_group("parser_advance_until_terminated"); - - for (name, data) in test_data.iter() { - if data.len() < 50 { - continue; - } // Skip small data - - group.bench_with_input( - BenchmarkId::new("terminate_early", name), - data, - |b, data| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = TerminatingPerformer::new(10); // Terminate after 10 characters - let _processed = parser - .advance_until_terminated(&mut performer, std_black_box(data)); - }); - }, - ); - } - - group.finish(); -} - -fn bench_utf8_scenarios(c: &mut Criterion) { - let mut group = c.benchmark_group("utf8_scenarios"); - - // Pure ASCII (should be fastest) - let ascii_data = "a".repeat(1000).into_bytes(); - group.bench_function("pure_ascii_1k", |b| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, std_black_box(&ascii_data)); - }); - }); - - // Pure UTF-8 (2-byte characters) - let utf8_2byte = "é".repeat(1000).into_bytes(); - group.bench_function("utf8_2byte_1k", |b| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, std_black_box(&utf8_2byte)); - }); - }); - - // Pure UTF-8 (3-byte characters) - let utf8_3byte = "中".repeat(1000).into_bytes(); - group.bench_function("utf8_3byte_1k", |b| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, std_black_box(&utf8_3byte)); - }); - }); - - // Pure UTF-8 (4-byte characters - emojis) - let utf8_4byte = "🦀".repeat(1000).into_bytes(); - group.bench_function("utf8_4byte_1k", |b| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, std_black_box(&utf8_4byte)); - }); - }); - - group.finish(); -} - -fn bench_real_world_scenarios(c: &mut Criterion) { - let mut group = c.benchmark_group("real_world"); - - // Simulate ls -la output with UTF-8 filenames - let ls_output = { - let mut data = Vec::new(); - for i in 0..100 { - data.extend_from_slice( - format!("drwxr-xr-x 2 user group 4096 Jan 1 12:00 📁folder_{i}\n") - .as_bytes(), - ); - data.extend_from_slice( - format!( - "-rw-r--r-- 1 user group 1024 Jan 1 12:00 📄file_{}_{}.txt\n", - i, "🦀" - ) - .as_bytes(), - ); - } - data - }; - - group.bench_function("ls_output", |b| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, std_black_box(&ls_output)); - }); - }); - - // Simulate git log output with UTF-8 commit messages - let git_log = { - let mut data = Vec::new(); - for i in 0..50 { - data.extend_from_slice( - format!("\x1b[33mcommit abc123{i}\x1b[0m\n").as_bytes(), - ); - data.extend_from_slice("Author: Developer 👨‍💻 \n".as_bytes()); - data.extend_from_slice("Date: Mon Jan 1 12:00:00 2024 +0000\n\n".as_bytes()); - data.extend_from_slice( - format!(" 🚀 Add feature {i} with 中文 support\n\n").as_bytes(), - ); - } - data - }; - - group.bench_function("git_log", |b| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, std_black_box(&git_log)); - }); - }); - - // Simulate cat on a source code file with UTF-8 comments - let source_code = { - let mut data = Vec::new(); - for i in 0..200 { - data.extend_from_slice( - format!("// This is a comment with UTF-8: 🦀 Rust code line {i}\n") - .as_bytes(), - ); - data.extend_from_slice( - format!("fn function_{i}() -> Result<(), Error> {{\n").as_bytes(), - ); - data.extend_from_slice(" println!(\"Hello, 世界! 🌍\");\n".as_bytes()); - data.extend_from_slice(b" Ok(())\n}\n\n"); - } - data - }; - - group.bench_function("source_code", |b| { - b.iter(|| { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, std_black_box(&source_code)); - }); - }); - - group.finish(); -} - -criterion_group!( - benches, - bench_parser_advance, - bench_parser_advance_chunked, - bench_parser_advance_until_terminated, - bench_utf8_scenarios, - bench_real_world_scenarios -); -criterion_main!(benches); diff --git a/copa/examples/advanced_benchmark.rs b/copa/examples/advanced_benchmark.rs deleted file mode 100644 index d3c50904..00000000 --- a/copa/examples/advanced_benchmark.rs +++ /dev/null @@ -1,424 +0,0 @@ -use copa::{Params, Parser, Perform}; -use std::time::Instant; - -struct NoOpPerformer; - -impl Perform for NoOpPerformer { - fn print(&mut self, _c: char) {} - fn execute(&mut self, _byte: u8) {} - fn hook( - &mut self, - _params: &Params, - _intermediates: &[u8], - _ignore: bool, - _action: char, - ) { - } - fn put(&mut self, _byte: u8) {} - fn unhook(&mut self) {} - fn osc_dispatch(&mut self, _params: &[&[u8]], _bell_terminated: bool) {} - fn csi_dispatch( - &mut self, - _params: &Params, - _intermediates: &[u8], - _ignore: bool, - _action: char, - ) { - } - fn esc_dispatch(&mut self, _intermediates: &[u8], _ignore: bool, _byte: u8) {} -} - -#[derive(Debug, Clone)] -struct BenchmarkResult { - name: String, - duration_ms: f64, - throughput_mbps: f64, - data_size: usize, - iterations: usize, -} - -impl BenchmarkResult { - fn new(name: &str, data: &[u8], iterations: usize) -> Self { - // Warm up - for _ in 0..10 { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, data); - } - - let start = Instant::now(); - - for _ in 0..iterations { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, data); - } - - let duration = start.elapsed(); - let duration_ms = duration.as_secs_f64() * 1000.0; - let throughput_mbps = - (data.len() * iterations) as f64 / duration.as_secs_f64() / 1_000_000.0; - - Self { - name: name.to_string(), - duration_ms, - throughput_mbps, - data_size: data.len(), - iterations, - } - } -} - -struct BenchmarkSuite { - results: Vec, -} - -impl BenchmarkSuite { - fn new() -> Self { - Self { - results: Vec::new(), - } - } - - fn add_test(&mut self, name: &str, data: &[u8], iterations: usize) { - let result = BenchmarkResult::new(name, data, iterations); - self.results.push(result); - } - - fn run_all_tests(&mut self) { - println!("Running comprehensive Copa parser benchmarks..."); - println!(); - - // Test 1: ASCII performance - let ascii_small = b"Hello, World! This is ASCII text.".repeat(10); - self.add_test("ASCII Small", &ascii_small, 10000); - - let ascii_large = b"Hello, World! This is ASCII text.".repeat(1000); - self.add_test("ASCII Large", &ascii_large, 1000); - - // Test 2: UTF-8 2-byte characters - let utf8_2byte = "café naïve résumé".repeat(100); - self.add_test("UTF-8 2-byte", utf8_2byte.as_bytes(), 5000); - - // Test 3: UTF-8 3-byte characters (CJK) - let utf8_3byte = "中文测试 日本語 한국어".repeat(100); - self.add_test("UTF-8 3-byte (CJK)", utf8_3byte.as_bytes(), 5000); - - // Test 4: UTF-8 4-byte characters (emojis) - let utf8_4byte = "🦀🚀🌟💫🎉✨🌍🔥".repeat(100); - self.add_test("UTF-8 4-byte (emoji)", utf8_4byte.as_bytes(), 5000); - - // Test 5: Mixed content - let mixed = "Hello 🌍! Welcome to Rust 🦀. This is a test with café, naïve, 中文, العربية, русский язык.".repeat(50); - self.add_test("Mixed UTF-8", mixed.as_bytes(), 3000); - - // Test 6: Terminal escape sequences - let escape_seq = b"\x1b[31mRed\x1b[0m \x1b[32mGreen\x1b[0m \x1b[34mBlue\x1b[0m \x1b[1mBold\x1b[0m".repeat(100); - self.add_test("Escape Sequences", &escape_seq, 3000); - - // Test 7: OSC sequences with UTF-8 - let osc_seq = - b"\x1b]2;Terminal Title: \xF0\x9F\x92\xBB Rust Terminal\x07".repeat(100); - self.add_test("OSC with UTF-8", &osc_seq, 3000); - - // Test 8: CSI sequences - let csi_seq = b"\x1b[1;32mBold Green\x1b[0m \x1b[4mUnderlined\x1b[0m \x1b[38;2;255;0;255mTruecolor\x1b[0m".repeat(100); - self.add_test("CSI Sequences", &csi_seq, 3000); - - // Test 9: Real-world ls output - let mut ls_output = Vec::new(); - for i in 0..100 { - ls_output.extend_from_slice( - format!("drwxr-xr-x 2 user group 4096 Jan 1 12:00 📁folder_{i}\n") - .as_bytes(), - ); - ls_output.extend_from_slice( - format!( - "-rw-r--r-- 1 user group 1024 Jan 1 12:00 📄file_{}_{}.txt\n", - i, "🦀" - ) - .as_bytes(), - ); - } - self.add_test("LS Output", &ls_output, 1000); - - // Test 10: Git log output - let mut git_log = Vec::new(); - for i in 0..50 { - git_log.extend_from_slice(format!( - "\x1b[33mcommit abc123{i}\x1b[0m\nAuthor: Dev 👨‍💻 \nDate: Mon Jan 1 12:00:00 2024\n\n 🚀 Feature {i} with 中文 support\n\n" - ).as_bytes()); - } - self.add_test("Git Log", &git_log, 1000); - - // Test 11: Source code with UTF-8 comments - let mut source_code = Vec::new(); - for i in 0..100 { - source_code.extend_from_slice(format!( - "// Comment with UTF-8: 🦀 Rust line {i}\nfn function_{i}() -> Result<(), Error> {{\n println!(\"Hello, 世界! 🌍\");\n Ok(())\n}}\n\n" - ).as_bytes()); - } - self.add_test("Source Code", &source_code, 1000); - - // Test 12: Chunked processing - let chunked_data = - "🎉🦀🚀 Rust is amazing! 中文测试 العربية русский язык 🌟✨💫".repeat(100); - let iterations = 1000; - - // Warm up - for _ in 0..10 { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - for chunk in chunked_data.as_bytes().chunks(16) { - parser.advance(&mut performer, chunk); - } - } - - let start = Instant::now(); - for _ in 0..iterations { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - for chunk in chunked_data.as_bytes().chunks(16) { - parser.advance(&mut performer, chunk); - } - } - let duration = start.elapsed(); - let duration_ms = duration.as_secs_f64() * 1000.0; - let throughput_mbps = (chunked_data.len() * iterations) as f64 - / duration.as_secs_f64() - / 1_000_000.0; - - self.results.push(BenchmarkResult { - name: "Chunked Processing".to_string(), - duration_ms, - throughput_mbps, - data_size: chunked_data.len(), - iterations, - }); - } - - fn print_results(&self) { - println!("{:-<90}", ""); - println!( - "{:25} | {:>10} | {:>10} | {:>12} | {:>8}", - "Test Case", "Time", "Throughput", "Data Size", "Iterations" - ); - println!("{:-<90}", ""); - - for result in &self.results { - println!( - "{:25} | {:8.2} ms | {:8.2} MB/s | {:6} bytes | {:4} iter", - result.name, - result.duration_ms, - result.throughput_mbps, - result.data_size, - result.iterations - ); - } - - println!("{:-<90}", ""); - } - - fn print_summary(&self) { - let total_throughput: f64 = self.results.iter().map(|r| r.throughput_mbps).sum(); - let avg_throughput = total_throughput / self.results.len() as f64; - let max_throughput = self - .results - .iter() - .map(|r| r.throughput_mbps) - .fold(0.0, f64::max); - let min_throughput = self - .results - .iter() - .map(|r| r.throughput_mbps) - .fold(f64::INFINITY, f64::min); - - println!(); - println!("Summary Statistics:"); - println!(" Average throughput: {avg_throughput:.2} MB/s"); - println!(" Maximum throughput: {max_throughput:.2} MB/s"); - println!(" Minimum throughput: {min_throughput:.2} MB/s"); - println!(" Total test cases: {}", self.results.len()); - - // Category analysis - let utf8_tests: Vec<_> = self - .results - .iter() - .filter(|r| { - r.name.contains("UTF-8") - || r.name.contains("emoji") - || r.name.contains("CJK") - }) - .collect(); - - let ascii_tests: Vec<_> = self - .results - .iter() - .filter(|r| r.name.contains("ASCII")) - .collect(); - - let real_world_tests: Vec<_> = self - .results - .iter() - .filter(|r| { - r.name.contains("LS") - || r.name.contains("Git") - || r.name.contains("Source") - }) - .collect(); - - println!(); - println!("Category Analysis:"); - - if !utf8_tests.is_empty() { - let utf8_avg: f64 = utf8_tests.iter().map(|r| r.throughput_mbps).sum::() - / utf8_tests.len() as f64; - println!( - " UTF-8 heavy workloads: {:.2} MB/s average ({} tests)", - utf8_avg, - utf8_tests.len() - ); - } - - if !ascii_tests.is_empty() { - let ascii_avg: f64 = - ascii_tests.iter().map(|r| r.throughput_mbps).sum::() - / ascii_tests.len() as f64; - println!( - " ASCII workloads: {:.2} MB/s average ({} tests)", - ascii_avg, - ascii_tests.len() - ); - } - - if !real_world_tests.is_empty() { - let real_world_avg: f64 = real_world_tests - .iter() - .map(|r| r.throughput_mbps) - .sum::() - / real_world_tests.len() as f64; - println!( - " Real-world scenarios: {:.2} MB/s average ({} tests)", - real_world_avg, - real_world_tests.len() - ); - } - } - - fn export_json(&self) -> String { - let mut json = String::from("{\n"); - json.push_str(&format!( - " \"timestamp\": \"{}\",\n", - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .unwrap() - .as_secs() - )); - json.push_str(&format!( - " \"implementation\": \"{}\",\n", - detect_implementation() - )); - json.push_str(&format!( - " \"rust_version\": \"{}\",\n", - std::env::var("RUSTC_VERSION").unwrap_or_else(|_| "unknown".to_string()) - )); - json.push_str(" \"results\": [\n"); - - for (i, result) in self.results.iter().enumerate() { - json.push_str(" {\n"); - json.push_str(&format!(" \"name\": \"{}\",\n", result.name)); - json.push_str(&format!( - " \"duration_ms\": {:.2},\n", - result.duration_ms - )); - json.push_str(&format!( - " \"throughput_mbps\": {:.2},\n", - result.throughput_mbps - )); - json.push_str(&format!(" \"data_size\": {},\n", result.data_size)); - json.push_str(&format!(" \"iterations\": {}\n", result.iterations)); - json.push_str(" }"); - if i < self.results.len() - 1 { - json.push(','); - } - json.push('\n'); - } - - json.push_str(" ]\n"); - json.push_str("}\n"); - json - } -} - -fn detect_implementation() -> &'static str { - // Try to detect if we're using simdutf8 by checking if it's available - match std::panic::catch_unwind(|| simdutf8::basic::from_utf8(b"test")) { - Ok(_) => "simdutf8", - Err(_) => "std", - } -} - -fn print_system_info() { - println!("Copa Parser Advanced Benchmark Suite"); - println!(); - - let implementation = detect_implementation(); - println!("Implementation: {implementation}"); - println!( - "Rust version: {}", - std::env::var("RUSTC_VERSION").unwrap_or_else(|_| "unknown".to_string()) - ); - println!( - "Target: {}", - std::env::var("TARGET").unwrap_or_else(|_| std::env::consts::ARCH.to_string()) - ); - - // Try to get CPU info - #[cfg(target_os = "macos")] - { - if let Ok(output) = std::process::Command::new("sysctl") - .args(["-n", "machdep.cpu.brand_string"]) - .output() - { - if let Ok(cpu_info) = String::from_utf8(output.stdout) { - println!("CPU: {}", cpu_info.trim()); - } - } - } - - #[cfg(target_os = "linux")] - { - if let Ok(content) = std::fs::read_to_string("/proc/cpuinfo") { - for line in content.lines() { - if line.starts_with("model name") { - if let Some(cpu_name) = line.split(':').nth(1) { - println!("CPU: {}", cpu_name.trim()); - break; - } - } - } - } - } - - println!(); -} - -fn main() { - print_system_info(); - - let mut suite = BenchmarkSuite::new(); - suite.run_all_tests(); - suite.print_results(); - suite.print_summary(); - - // Export results to JSON for comparison - let json_output = suite.export_json(); - let filename = format!("copa_benchmark_{}.json", detect_implementation()); - - if let Err(e) = std::fs::write(&filename, json_output) { - eprintln!("Warning: Could not write results to {filename}: {e}"); - } else { - println!(); - println!("Results exported to: {filename}"); - } -} diff --git a/copa/examples/benchmark_comparison.rs b/copa/examples/benchmark_comparison.rs deleted file mode 100644 index a978a3c5..00000000 --- a/copa/examples/benchmark_comparison.rs +++ /dev/null @@ -1,288 +0,0 @@ -use copa::{Params, Parser, Perform}; -use std::time::Instant; - -struct NoOpPerformer; - -impl Perform for NoOpPerformer { - fn print(&mut self, _c: char) {} - fn execute(&mut self, _byte: u8) {} - fn hook( - &mut self, - _params: &Params, - _intermediates: &[u8], - _ignore: bool, - _action: char, - ) { - } - fn put(&mut self, _byte: u8) {} - fn unhook(&mut self) {} - fn osc_dispatch(&mut self, _params: &[&[u8]], _bell_terminated: bool) {} - fn csi_dispatch( - &mut self, - _params: &Params, - _intermediates: &[u8], - _ignore: bool, - _action: char, - ) { - } - fn esc_dispatch(&mut self, _intermediates: &[u8], _ignore: bool, _byte: u8) {} -} - -#[derive(Debug)] -struct BenchmarkResult { - name: String, - duration_ms: f64, - throughput_mbps: f64, - data_size: usize, - iterations: usize, -} - -impl BenchmarkResult { - fn new(name: &str, data: &[u8], iterations: usize) -> Self { - let start = Instant::now(); - - for _ in 0..iterations { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - parser.advance(&mut performer, data); - } - - let duration = start.elapsed(); - let duration_ms = duration.as_secs_f64() * 1000.0; - let throughput_mbps = - (data.len() * iterations) as f64 / duration.as_secs_f64() / 1_000_000.0; - - Self { - name: name.to_string(), - duration_ms, - throughput_mbps, - data_size: data.len(), - iterations, - } - } - - fn print(&self) { - println!( - "{:25} | {:8.2} ms | {:8.2} MB/s | {:6} bytes | {:4} iter", - self.name, - self.duration_ms, - self.throughput_mbps, - self.data_size, - self.iterations - ); - } -} - -fn generate_test_data() -> Vec<(&'static str, Vec, usize)> { - vec![ - // (name, data, iterations) - ("ASCII Small", b"Hello, World! This is ASCII text.".repeat(10), 10000), - ("ASCII Large", b"Hello, World! This is ASCII text.".repeat(1000), 1000), - ("UTF-8 2-byte", "café naïve résumé".repeat(100).into_bytes(), 5000), - ("UTF-8 3-byte (CJK)", "中文测试 日本語 한국어".repeat(100).into_bytes(), 5000), - ("UTF-8 4-byte (emoji)", "🦀🚀🌟💫🎉✨🌍🔥".repeat(100).into_bytes(), 5000), - ("Mixed UTF-8", "Hello 🌍! Welcome to Rust 🦀. This is a test with café, naïve, 中文, العربية, русский язык.".repeat(50).into_bytes(), 3000), - ("Escape Sequences", b"\x1b[31mRed\x1b[0m \x1b[32mGreen\x1b[0m \x1b[34mBlue\x1b[0m \x1b[1mBold\x1b[0m".repeat(100), 3000), - ("OSC with UTF-8", b"\x1b]2;Terminal Title: \xF0\x9F\x92\xBB Rust Terminal\x07".repeat(100), 3000), - ("CSI Sequences", b"\x1b[1;32mBold Green\x1b[0m \x1b[4mUnderlined\x1b[0m \x1b[38;2;255;0;255mTruecolor\x1b[0m".repeat(100), 3000), - ("LS Output", { - let mut data = Vec::new(); - for i in 0..100 { - data.extend_from_slice(format!( - "drwxr-xr-x 2 user group 4096 Jan 1 12:00 📁folder_{}\n-rw-r--r-- 1 user group 1024 Jan 1 12:00 📄file_{}_{}.txt\n", - i, i, "🦀" - ).as_bytes()); - } - data - }, 1000), - ("Git Log", { - let mut data = Vec::new(); - for i in 0..50 { - data.extend_from_slice(format!( - "\x1b[33mcommit abc123{i}\x1b[0m\nAuthor: Dev 👨‍💻 \nDate: Mon Jan 1 12:00:00 2024\n\n 🚀 Feature {i} with 中文 support\n\n" - ).as_bytes()); - } - data - }, 1000), - ("Source Code", { - let mut data = Vec::new(); - for i in 0..100 { - data.extend_from_slice(format!( - "// Comment with UTF-8: 🦀 Rust line {i}\nfn function_{i}() -> Result<(), Error> {{\n println!(\"Hello, 世界! 🌍\");\n Ok(())\n}}\n\n" - ).as_bytes()); - } - data - }, 1000), - ] -} - -fn run_chunked_test() -> BenchmarkResult { - let chunked_data = - "🎉🦀🚀 Rust is amazing! 中文测试 العربية русский язык 🌟✨💫".repeat(100); - let iterations = 1000; - let start = Instant::now(); - - for _ in 0..iterations { - let mut parser = Parser::new(); - let mut performer = NoOpPerformer; - // Process in small chunks like real terminal input - for chunk in chunked_data.as_bytes().chunks(16) { - parser.advance(&mut performer, chunk); - } - } - - let duration = start.elapsed(); - let duration_ms = duration.as_secs_f64() * 1000.0; - let throughput_mbps = - (chunked_data.len() * iterations) as f64 / duration.as_secs_f64() / 1_000_000.0; - - BenchmarkResult { - name: "Chunked Processing".to_string(), - duration_ms, - throughput_mbps, - data_size: chunked_data.len(), - iterations, - } -} - -fn print_system_info() { - println!("Copa Parser Performance Benchmark"); - - // Try to detect if we're using simdutf8 or std - let implementation = - match std::panic::catch_unwind(|| simdutf8::basic::from_utf8(b"test")) { - Ok(_) => "simdutf8 (SIMD-accelerated)", - Err(_) => "std::str (standard library)", - }; - - println!("Implementation: {implementation}"); - println!( - "Rust version: {}", - std::env::var("RUSTC_VERSION").unwrap_or_else(|_| "unknown".to_string()) - ); - println!( - "Target: {}", - std::env::var("TARGET").unwrap_or_else(|_| std::env::consts::ARCH.to_string()) - ); - - // Try to get CPU info - #[cfg(target_os = "macos")] - { - if let Ok(output) = std::process::Command::new("sysctl") - .args(["-n", "machdep.cpu.brand_string"]) - .output() - { - if let Ok(cpu_info) = String::from_utf8(output.stdout) { - println!("CPU: {}", cpu_info.trim()); - } - } - } - - #[cfg(target_os = "linux")] - { - if let Ok(content) = std::fs::read_to_string("/proc/cpuinfo") { - for line in content.lines() { - if line.starts_with("model name") { - if let Some(cpu_name) = line.split(':').nth(1) { - println!("CPU: {}", cpu_name.trim()); - break; - } - } - } - } - } - - println!(); -} - -fn main() { - print_system_info(); - - let test_data = generate_test_data(); - - println!("{:-<90}", ""); - println!( - "{:25} | {:>10} | {:>10} | {:>12} | {:>8}", - "Test Case", "Time", "Throughput", "Data Size", "Iterations" - ); - println!("{:-<90}", ""); - - let mut results = Vec::new(); - - // Run chunked test first - let chunked_result = run_chunked_test(); - chunked_result.print(); - results.push(chunked_result); - - // Run all other tests - for (name, data, iterations) in test_data { - let result = BenchmarkResult::new(name, &data, iterations); - result.print(); - results.push(result); - } - - println!("{:-<90}", ""); - - // Calculate summary statistics - let total_throughput: f64 = results.iter().map(|r| r.throughput_mbps).sum(); - let avg_throughput = total_throughput / results.len() as f64; - let max_throughput = results - .iter() - .map(|r| r.throughput_mbps) - .fold(0.0, f64::max); - let min_throughput = results - .iter() - .map(|r| r.throughput_mbps) - .fold(f64::INFINITY, f64::min); - - println!(); - println!("Summary Statistics:"); - println!(" Average throughput: {avg_throughput:.2} MB/s"); - println!(" Maximum throughput: {max_throughput:.2} MB/s"); - println!(" Minimum throughput: {min_throughput:.2} MB/s"); - println!(" Total test cases: {}", results.len()); - - println!(); - println!("Performance Analysis:"); - - // Categorize results - let utf8_tests: Vec<_> = results - .iter() - .filter(|r| { - r.name.contains("UTF-8") || r.name.contains("emoji") || r.name.contains("CJK") - }) - .collect(); - - let ascii_tests: Vec<_> = results - .iter() - .filter(|r| r.name.contains("ASCII")) - .collect(); - - let real_world_tests: Vec<_> = results - .iter() - .filter(|r| { - r.name.contains("LS") || r.name.contains("Git") || r.name.contains("Source") - }) - .collect(); - - if !utf8_tests.is_empty() { - let utf8_avg: f64 = utf8_tests.iter().map(|r| r.throughput_mbps).sum::() - / utf8_tests.len() as f64; - println!(" UTF-8 heavy workloads: {utf8_avg:.2} MB/s average"); - } - - if !ascii_tests.is_empty() { - let ascii_avg: f64 = ascii_tests.iter().map(|r| r.throughput_mbps).sum::() - / ascii_tests.len() as f64; - println!(" ASCII workloads: {ascii_avg:.2} MB/s average"); - } - - if !real_world_tests.is_empty() { - let real_world_avg: f64 = real_world_tests - .iter() - .map(|r| r.throughput_mbps) - .sum::() - / real_world_tests.len() as f64; - println!(" Real-world scenarios: {real_world_avg:.2} MB/s average"); - } -} diff --git a/copa/examples/parselog.rs b/copa/examples/parselog.rs deleted file mode 100644 index 2d2f70e9..00000000 --- a/copa/examples/parselog.rs +++ /dev/null @@ -1,76 +0,0 @@ -//! Parse input from stdin and log actions on stdout -use std::io::{self, Read}; - -use copa::{Params, Parser, Perform}; - -/// A type implementing Perform that just logs actions -struct Log; - -impl Perform for Log { - fn print(&mut self, c: char) { - println!("[print] {c:?}"); - } - - fn execute(&mut self, byte: u8) { - println!("[execute] {byte:02x}"); - } - - fn hook(&mut self, params: &Params, intermediates: &[u8], ignore: bool, c: char) { - println!( - "[hook] params={params:?}, intermediates={intermediates:?}, ignore={ignore:?}, char={c:?}" - ); - } - - fn put(&mut self, byte: u8) { - println!("[put] {byte:02x}"); - } - - fn unhook(&mut self) { - println!("[unhook]"); - } - - fn osc_dispatch(&mut self, params: &[&[u8]], bell_terminated: bool) { - println!("[osc_dispatch] params={params:?} bell_terminated={bell_terminated}"); - } - - fn csi_dispatch( - &mut self, - params: &Params, - intermediates: &[u8], - ignore: bool, - c: char, - ) { - println!( - "[csi_dispatch] params={params:#?}, intermediates={intermediates:?}, ignore={ignore:?}, char={c:?}" - ); - } - - fn esc_dispatch(&mut self, intermediates: &[u8], ignore: bool, byte: u8) { - println!( - "[esc_dispatch] intermediates={intermediates:?}, ignore={ignore:?}, byte={byte:02x}" - ); - } -} - -fn main() { - let input = io::stdin(); - let mut handle = input.lock(); - - let mut statemachine = Parser::new(); - let mut performer = Log; - - let mut buf = [0; 2048]; - - loop { - match handle.read(&mut buf) { - Ok(0) => break, - Ok(_n) => { - statemachine.advance(&mut performer, &buf); - } - Err(err) => { - println!("err: {err}"); - break; - } - } - } -} diff --git a/copa/src/definitions.rs b/copa/src/definitions.rs deleted file mode 100644 index 7f0e8894..00000000 --- a/copa/src/definitions.rs +++ /dev/null @@ -1,101 +0,0 @@ -use core::mem; - -#[allow(dead_code)] -#[repr(u8)] -#[derive(PartialEq, Eq, Debug, Default, Copy, Clone)] -pub enum State { - CsiEntry, - CsiIgnore, - CsiIntermediate, - CsiParam, - DcsEntry, - DcsIgnore, - DcsIntermediate, - DcsParam, - DcsPassthrough, - Escape, - EscapeIntermediate, - OscString, - SosString, - ApcString, - PmString, - Anywhere, - #[default] - Ground, -} - -// NOTE: Removing the unused actions prefixed with `_` will reduce performance. -#[allow(dead_code)] -#[repr(u8)] -#[derive(PartialEq, Eq, Debug, Clone, Copy)] -pub enum Action { - None, - _Clear, - Collect, - CsiDispatch, - EscDispatch, - Execute, - _Hook, - _Ignore, - _OscEnd, - OscPut, - _OscStart, - Param, - _Print, - Put, - _Unhook, -} - -/// Unpack a u8 into a State and Action -/// -/// The implementation of this assumes that there are *precisely* 16 variants -/// for both Action and State. Furthermore, it assumes that the enums are -/// tag-only; that is, there is no data in any variant. -/// -/// Bad things will happen if those invariants are violated. -#[inline(always)] -pub fn unpack(delta: u8) -> (State, Action) { - unsafe { - ( - // State is stored in bottom 4 bits - mem::transmute::(delta & 0x0F), - // Action is stored in top 4 bits - mem::transmute::(delta >> 4), - ) - } -} - -#[inline(always)] -pub const fn pack(state: State, action: Action) -> u8 { - (action as u8) << 4 | state as u8 -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn unpack_state_action() { - match unpack(0xEE) { - (State::Ground, Action::_Unhook) => (), - _ => panic!("unpack failed"), - } - - match unpack(0x0E) { - (State::Ground, Action::None) => (), - _ => panic!("unpack failed"), - } - - match unpack(0xE0) { - (State::CsiEntry, Action::_Unhook) => (), - _ => panic!("unpack failed"), - } - } - - #[test] - fn pack_state_action() { - assert_eq!(pack(State::Ground, Action::_Unhook), 0xEE); - assert_eq!(pack(State::Ground, Action::None), 0x0E); - assert_eq!(pack(State::CsiEntry, Action::_Unhook), 0xE0); - } -} diff --git a/copa/src/table.rs b/copa/src/table.rs deleted file mode 100644 index 5ad34204..00000000 --- a/copa/src/table.rs +++ /dev/null @@ -1,188 +0,0 @@ -use rio_proc_macros::generate_state_changes; - -/// This is the state change table. It's indexed first by current state and then -/// by the next character in the pty stream. -use crate::definitions::{pack, Action, State}; - -// Generate state changes at compile-time -pub const STATE_CHANGES: [[u8; 256]; 13] = state_changes(); -generate_state_changes!(state_changes, { - Escape { - 0x00..=0x17 => (Anywhere, Execute), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, Execute), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, Execute), - 0x7f => (Anywhere, None), - 0x20..=0x2f => (EscapeIntermediate, Collect), - 0x30..=0x4f => (Ground, EscDispatch), - 0x51..=0x57 => (Ground, EscDispatch), - 0x59 => (Ground, EscDispatch), - 0x5a => (Ground, EscDispatch), - 0x5c => (Ground, EscDispatch), - 0x60..=0x7e => (Ground, EscDispatch), - 0x5b => (CsiEntry, None), - 0x5d => (OscString, None), - 0x50 => (DcsEntry, None), - 0x58 => (SosPmApcString, None), - 0x5e => (SosPmApcString, None), - 0x5f => (SosPmApcString, None), - }, - - EscapeIntermediate { - 0x00..=0x17 => (Anywhere, Execute), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, Execute), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, Execute), - 0x20..=0x2f => (Anywhere, Collect), - 0x7f => (Anywhere, None), - 0x30..=0x7e => (Ground, EscDispatch), - }, - - CsiEntry { - 0x00..=0x17 => (Anywhere, Execute), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, Execute), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, Execute), - 0x7f => (Anywhere, None), - 0x20..=0x2f => (CsiIntermediate, Collect), - 0x30..=0x39 => (CsiParam, Param), - 0x3a..=0x3b => (CsiParam, Param), - 0x3c..=0x3f => (CsiParam, Collect), - 0x40..=0x7e => (Ground, CsiDispatch), - }, - - CsiIgnore { - 0x00..=0x17 => (Anywhere, Execute), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, Execute), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, Execute), - 0x20..=0x3f => (Anywhere, None), - 0x7f => (Anywhere, None), - 0x40..=0x7e => (Ground, None), - }, - - CsiParam { - 0x00..=0x17 => (Anywhere, Execute), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, Execute), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, Execute), - 0x30..=0x39 => (Anywhere, Param), - 0x3a..=0x3b => (Anywhere, Param), - 0x7f => (Anywhere, None), - 0x3c..=0x3f => (CsiIgnore, None), - 0x20..=0x2f => (CsiIntermediate, Collect), - 0x40..=0x7e => (Ground, CsiDispatch), - }, - - CsiIntermediate { - 0x00..=0x17 => (Anywhere, Execute), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, Execute), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, Execute), - 0x20..=0x2f => (Anywhere, Collect), - 0x7f => (Anywhere, None), - 0x30..=0x3f => (CsiIgnore, None), - 0x40..=0x7e => (Ground, CsiDispatch), - }, - - DcsEntry { - 0x00..=0x17 => (Anywhere, None), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, None), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, None), - 0x7f => (Anywhere, None), - 0x20..=0x2f => (DcsIntermediate, Collect), - 0x30..=0x39 => (DcsParam, Param), - 0x3a..=0x3b => (DcsParam, Param), - 0x3c..=0x3f => (DcsParam, Collect), - 0x40..=0x7e => (DcsPassthrough, None), - }, - - DcsIntermediate { - 0x00..=0x17 => (Anywhere, None), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, None), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, None), - 0x20..=0x2f => (Anywhere, Collect), - 0x7f => (Anywhere, None), - 0x30..=0x3f => (DcsIgnore, None), - 0x40..=0x7e => (DcsPassthrough, None), - }, - - DcsIgnore { - 0x00..=0x17 => (Anywhere, None), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, None), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, None), - 0x20..=0x7f => (Anywhere, None), - 0x9c => (Ground, None), - }, - - DcsParam { - 0x00..=0x17 => (Anywhere, None), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, None), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, None), - 0x30..=0x39 => (Anywhere, Param), - 0x3a..=0x3b => (Anywhere, Param), - 0x7f => (Anywhere, None), - 0x3c..=0x3f => (DcsIgnore, None), - 0x20..=0x2f => (DcsIntermediate, Collect), - 0x40..=0x7e => (DcsPassthrough, None), - }, - - DcsPassthrough { - 0x00..=0x17 => (Anywhere, Put), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, Put), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, Put), - 0x20..=0x7e => (Anywhere, Put), - 0x7f => (Anywhere, None), - 0x9c => (Ground, None), - }, - - SosPmApcString { - 0x00..=0x17 => (Anywhere, None), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, None), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, None), - 0x20..=0x7f => (Anywhere, None), - 0x9c => (Ground, None), - }, - - OscString { - 0x00..=0x06 => (Anywhere, None), - 0x07 => (Ground, None), - 0x08..=0x17 => (Anywhere, None), - 0x18 => (Ground, Execute), - 0x19 => (Anywhere, None), - 0x1a => (Ground, Execute), - 0x1b => (Escape, None), - 0x1c..=0x1f => (Anywhere, None), - 0x20..=0xff => (Anywhere, OscPut), - } -}); diff --git a/copa/tests/demo.vte b/copa/tests/demo.vte deleted file mode 100644 index ed62b355..00000000 --- a/copa/tests/demo.vte +++ /dev/null @@ -1,2052 +0,0 @@ -Test -RED ON GREEN -]52;c;Y2xpcGJvYXJkIHRlc3Q=];;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;some random text;  what - - -P0;1|17/ab� - ]2;echo '¯\_(ツ)_/¯' && sleep 1 -REEEEEED -Test -RED ON GREEN -]52;c;Y2xpcGJvYXJkIHRlc3Q=];;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;some random text;  what - - -P0;1|17/ab� - ]2;echo '¯\_(ツ)_/¯' && sleep 1 -REEEEEEEEEEEEED -[6 q -[0 q - - - -0 -1 -2 -3 -4 -5 -6 -7 -8 -9 -10 -11 -12 -13 -14 -15 -16 -17 -18 -19 -20 -21 -22 -23 -24 -25 -26 -27 -28 -29 -30 -31 -32 -33 -34 -35 -36 -37 -38 -39 -40 -41 -42 -43 -44 -45 -46 -47 -48 -49 -50 -51 -52 -53 -54 -55 -56 -57 -58 -59 -60 -61 -62 -63 -64 -65 -66 -67 -68 -69 -70 -71 -72 -73 -74 -75 -76 -77 -78 -79 -80 -81 -82 -83 -84 -85 -86 -87 -88 -89 -90 -91 -92 -93 -94 -95 -96 -97 -98 -99 -100 -101 -102 -103 -104 -105 -106 -107 -108 -109 -110 -111 -112 -113 -114 -115 -116 -117 -118 -119 -120 -121 -122 -123 -124 -125 -126 -127 -128 -129 -130 -131 -132 -133 -134 -135 -136 -137 -138 -139 -140 -141 -142 -143 -144 -145 -146 -147 -148 -149 -150 -151 -152 -153 -154 -155 -156 -157 -158 -159 -160 -161 -162 -163 -164 -165 -166 -167 -168 -169 -170 -171 -172 -173 -174 -175 -176 -177 -178 -179 -180 -181 -182 -183 -184 -185 -186 -187 -188 -189 -190 -191 -192 -193 -194 -195 -196 -197 -198 -199 -200 -201 -202 -203 -204 -205 -206 -207 -208 -209 -210 -211 -212 -213 -214 -215 -216 -217 -218 -219 -220 -221 -222 -223 -224 -225 -226 -227 -228 -229 -230 -231 -232 -233 -234 -235 -236 -237 -238 -239 -240 -241 -242 -243 -244 -245 -246 -247 -248 -249 -250 -251 -252 -253 -254 -255 -256 -257 -258 -259 -260 -261 -262 -263 -264 -265 -266 -267 -268 -269 -270 -271 -272 -273 -274 -275 -276 -277 -278 -279 -280 -281 -282 -283 -284 -285 -286 -287 -288 -289 -290 -291 -292 -293 -294 -295 -296 -297 -298 -299 -300 -301 -302 -303 -304 -305 -306 -307 -308 -309 -310 -311 -312 -313 -314 -315 -316 -317 -318 -319 -320 -321 -322 -323 -324 -325 -326 -327 -328 -329 -330 -331 -332 -333 -334 -335 -336 -337 -338 -339 -340 -341 -342 -343 -344 -345 -346 -347 -348 -349 -350 -351 -352 -353 -354 -355 -356 -357 -358 -359 -360 -361 -362 -363 -364 -365 -366 -367 -368 -369 -370 -371 -372 -373 -374 -375 -376 -377 -378 -379 -380 -381 -382 -383 -384 -385 -386 -387 -388 -389 -390 -391 -392 -393 -394 -395 -396 -397 -398 -399 -400 -401 -402 -403 -404 -405 -406 -407 -408 -409 -410 -411 -412 -413 -414 -415 -416 -417 -418 -419 -420 -421 -422 -423 -424 -425 -426 -427 -428 -429 -430 -431 -432 -433 -434 -435 -436 -437 -438 -439 -440 -441 -442 -443 -444 -445 -446 -447 -448 -449 -450 -451 -452 -453 -454 -455 -456 -457 -458 -459 -460 -461 -462 -463 -464 -465 -466 -467 -468 -469 -470 -471 -472 -473 -474 -475 -476 -477 -478 -479 -480 -481 -482 -483 -484 -485 -486 -487 -488 -489 -490 -491 -492 -493 -494 -495 -496 -497 -498 -499 -500 -501 -502 -503 -504 -505 -506 -507 -508 -509 -510 -511 -512 -513 -514 -515 -516 -517 -518 -519 -520 -521 -522 -523 -524 -525 -526 -527 -528 -529 -530 -531 -532 -533 -534 -535 -536 -537 -538 -539 -540 -541 -542 -543 -544 -545 -546 -547 -548 -549 -550 -551 -552 -553 -554 -555 -556 -557 -558 -559 -560 -561 -562 -563 -564 -565 -566 -567 -568 -569 -570 -571 -572 -573 -574 -575 -576 -577 -578 -579 -580 -581 -582 -583 -584 -585 -586 -587 -588 -589 -590 -591 -592 -593 -594 -595 -596 -597 -598 -599 -600 -601 -602 -603 -604 -605 -606 -607 -608 -609 -610 -611 -612 -613 -614 -615 -616 -617 -618 -619 -620 -621 -622 -623 -624 -625 -626 -627 -628 -629 -630 -631 -632 -633 -634 -635 -636 -637 -638 -639 -640 -641 -642 -643 -644 -645 -646 -647 -648 -649 -650 -651 -652 -653 -654 -655 -656 -657 -658 -659 -660 -661 -662 -663 -664 -665 -666 -667 -668 -669 -670 -671 -672 -673 -674 -675 -676 -677 -678 -679 -680 -681 -682 -683 -684 -685 -686 -687 -688 -689 -690 -691 -692 -693 -694 -695 -696 -697 -698 -699 -700 -701 -702 -703 -704 -705 -706 -707 -708 -709 -710 -711 -712 -713 -714 -715 -716 -717 -718 -719 -720 -721 -722 -723 -724 -725 -726 -727 -728 -729 -730 -731 -732 -733 -734 -735 -736 -737 -738 -739 -740 -741 -742 -743 -744 -745 -746 -747 -748 -749 -750 -751 -752 -753 -754 -755 -756 -757 -758 -759 -760 -761 -762 -763 -764 -765 -766 -767 -768 -769 -770 -771 -772 -773 -774 -775 -776 -777 -778 -779 -780 -781 -782 -783 -784 -785 -786 -787 -788 -789 -790 -791 -792 -793 -794 -795 -796 -797 -798 -799 -800 -801 -802 -803 -804 -805 -806 -807 -808 -809 -810 -811 -812 -813 -814 -815 -816 -817 -818 -819 -820 -821 -822 -823 -824 -825 -826 -827 -828 -829 -830 -831 -832 -833 -834 -835 -836 -837 -838 -839 -840 -841 -842 -843 -844 -845 -846 -847 -848 -849 -850 -851 -852 -853 -854 -855 -856 -857 -858 -859 -860 -861 -862 -863 -864 -865 -866 -867 -868 -869 -870 -871 -872 -873 -874 -875 -876 -877 -878 -879 -880 -881 -882 -883 -884 -885 -886 -887 -888 -889 -890 -891 -892 -893 -894 -895 -896 -897 -898 -899 -900 -901 -902 -903 -904 -905 -906 -907 -908 -909 -910 -911 -912 -913 -914 -915 -916 -917 -918 -919 -920 -921 -922 -923 -924 -925 -926 -927 -928 -929 -930 -931 -932 -933 -934 -935 -936 -937 -938 -939 -940 -941 -942 -943 -944 -945 -946 -947 -948 -949 -950 -951 -952 -953 -954 -955 -956 -957 -958 -959 -960 -961 -962 -963 -964 -965 -966 -967 -968 -969 -970 -971 -972 -973 -974 -975 -976 -977 -978 -979 -980 -981 -982 -983 -984 -985 -986 -987 -988 -989 -990 -991 -992 -993 -994 -995 -996 -997 -998 -999 -1000 - -[0 -1 -2 -3 -4 -5 -6 -7 -8 -9 -10 -11 -12 -13 -14 -15 -16 -17 -18 -19 -20 -21 -22 -23 -24 -25 -26 -27 -28 -29 -30 -31 -32 -33 -34 -35 -36 -37 -38 -39 -40 -41 -42 -43 -44 -45 -46 -47 -48 -49 -50 -51 -52 -53 -54 -55 -56 -57 -58 -59 -60 -61 -62 -63 -64 -65 -66 -67 -68 -69 -70 -71 -72 -73 -74 -75 -76 -77 -78 -79 -80 -81 -82 -83 -84 -85 -86 -87 -88 -89 -90 -91 -92 -93 -94 -95 -96 -97 -98 -99 -100 -101 -102 -103 -104 -105 -106 -107 -108 -109 -110 -111 -112 -113 -114 -115 -116 -117 -118 -119 -120 -121 -122 -123 -124 -125 -126 -127 -128 -129 -130 -131 -132 -133 -134 -135 -136 -137 -138 -139 -140 -141 -142 -143 -144 -145 -146 -147 -148 -149 -150 -151 -152 -153 -154 -155 -156 -157 -158 -159 -160 -161 -162 -163 -164 -165 -166 -167 -168 -169 -170 -171 -172 -173 -174 -175 -176 -177 -178 -179 -180 -181 -182 -183 -184 -185 -186 -187 -188 -189 -190 -191 -192 -193 -194 -195 -196 -197 -198 -199 -200 -201 -202 -203 -204 -205 -206 -207 -208 -209 -210 -211 -212 -213 -214 -215 -216 -217 -218 -219 -220 -221 -222 -223 -224 -225 -226 -227 -228 -229 -230 -231 -232 -233 -234 -235 -236 -237 -238 -239 -240 -241 -242 -243 -244 -245 -246 -247 -248 -249 -250 -251 -252 -253 -254 -255 -256 -257 -258 -259 -260 -261 -262 -263 -264 -265 -266 -267 -268 -269 -270 -271 -272 -273 -274 -275 -276 -277 -278 -279 -280 -281 -282 -283 -284 -285 -286 -287 -288 -289 -290 -291 -292 -293 -294 -295 -296 -297 -298 -299 -300 -301 -302 -303 -304 -305 -306 -307 -308 -309 -310 -311 -312 -313 -314 -315 -316 -317 -318 -319 -320 -321 -322 -323 -324 -325 -326 -327 -328 -329 -330 -331 -332 -333 -334 -335 -336 -337 -338 -339 -340 -341 -342 -343 -344 -345 -346 -347 -348 -349 -350 -351 -352 -353 -354 -355 -356 -357 -358 -359 -360 -361 -362 -363 -364 -365 -366 -367 -368 -369 -370 -371 -372 -373 -374 -375 -376 -377 -378 -379 -380 -381 -382 -383 -384 -385 -386 -387 -388 -389 -390 -391 -392 -393 -394 -395 -396 -397 -398 -399 -400 -401 -402 -403 -404 -405 -406 -407 -408 -409 -410 -411 -412 -413 -414 -415 -416 -417 -418 -419 -420 -421 -422 -423 -424 -425 -426 -427 -428 -429 -430 -431 -432 -433 -434 -435 -436 -437 -438 -439 -440 -441 -442 -443 -444 -445 -446 -447 -448 -449 -450 -451 -452 -453 -454 -455 -456 -457 -458 -459 -460 -461 -462 -463 -464 -465 -466 -467 -468 -469 -470 -471 -472 -473 -474 -475 -476 -477 -478 -479 -480 -481 -482 -483 -484 -485 -486 -487 -488 -489 -490 -491 -492 -493 -494 -495 -496 -497 -498 -499 -500 -501 -502 -503 -504 -505 -506 -507 -508 -509 -510 -511 -512 -513 -514 -515 -516 -517 -518 -519 -520 -521 -522 -523 -524 -525 -526 -527 -528 -529 -530 -531 -532 -533 -534 -535 -536 -537 -538 -539 -540 -541 -542 -543 -544 -545 -546 -547 -548 -549 -550 -551 -552 -553 -554 -555 -556 -557 -558 -559 -560 -561 -562 -563 -564 -565 -566 -567 -568 -569 -570 -571 -572 -573 -574 -575 -576 -577 -578 -579 -580 -581 -582 -583 -584 -585 -586 -587 -588 -589 -590 -591 -592 -593 -594 -595 -596 -597 -598 -599 -600 -601 -602 -603 -604 -605 -606 -607 -608 -609 -610 -611 -612 -613 -614 -615 -616 -617 -618 -619 -620 -621 -622 -623 -624 -625 -626 -627 -628 -629 -630 -631 -632 -633 -634 -635 -636 -637 -638 -639 -640 -641 -642 -643 -644 -645 -646 -647 -648 -649 -650 -651 -652 -653 -654 -655 -656 -657 -658 -659 -660 -661 -662 -663 -664 -665 -666 -667 -668 -669 -670 -671 -672 -673 -674 -675 -676 -677 -678 -679 -680 -681 -682 -683 -684 -685 -686 -687 -688 -689 -690 -691 -692 -693 -694 -695 -696 -697 -698 -699 -700 -701 -702 -703 -704 -705 -706 -707 -708 -709 -710 -711 -712 -713 -714 -715 -716 -717 -718 -719 -720 -721 -722 -723 -724 -725 -726 -727 -728 -729 -730 -731 -732 -733 -734 -735 -736 -737 -738 -739 -740 -741 -742 -743 -744 -745 -746 -747 -748 -749 -750 -751 -752 -753 -754 -755 -756 -757 -758 -759 -760 -761 -762 -763 -764 -765 -766 -767 -768 -769 -770 -771 -772 -773 -774 -775 -776 -777 -778 -779 -780 -781 -782 -783 -784 -785 -786 -787 -788 -789 -790 -791 -792 -793 -794 -795 -796 -797 -798 -799 -800 -801 -802 -803 -804 -805 -806 -807 -808 -809 -810 -811 -812 -813 -814 -815 -816 -817 -818 -819 -820 -821 -822 -823 -824 -825 -826 -827 -828 -829 -830 -831 -832 -833 -834 -835 -836 -837 -838 -839 -840 -841 -842 -843 -844 -845 -846 -847 -848 -849 -850 -851 -852 -853 -854 -855 -856 -857 -858 -859 -860 -861 -862 -863 -864 -865 -866 -867 -868 -869 -870 -871 -872 -873 -874 -875 -876 -877 -878 -879 -880 -881 -882 -883 -884 -885 -886 -887 -888 -889 -890 -891 -892 -893 -894 -895 -896 -897 -898 -899 -900 -901 -902 -903 -904 -905 -906 -907 -908 -909 -910 -911 -912 -913 -914 -915 -916 -917 -918 -919 -920 -921 -922 -923 -924 -925 -926 -927 -928 -929 -930 -931 -932 -933 -934 -935 -936 -937 -938 -939 -940 -941 -942 -943 -944 -945 -946 -947 -948 -949 -950 -951 -952 -953 -954 -955 -956 -957 -958 -959 -960 -961 -962 -963 -964 -965 -966 -967 -968 -969 -970 -971 -972 -973 -974 -975 -976 -977 -978 -979 -980 -981 -982 -983 -984 -985 -986 -987 -988 -989 -990 -991 -992 -993 -994 -995 -996 -997 -998 -999 -1000 - -Hello, World -Oops -Z -汉字漢字汉字漢字汉字漢字汉字漢字汉字漢字汉字漢字汉字漢字汉字漢字 - - - --[ \eAAAAAAABBBBBBBBBBBBCCCCCCCCCCCCCCCCCCCCDDDDDDDDDDDDDDDDDDEEEEEEEEEFFFFFFFFFGGGGGGG -�S�A8SyD�4+�ĜPVLx� �y����ij�)]��9��Vmb�Ю+~=�K�.�t�I zep+�d�ۨX��120�;�#�����r=:�Bʾ>�v�ox�Y̊曮Bk�����.�l�&��=���^�"h�l� 9˖��(��ɇ�zX�����R�V5�0���x>�%1�]/��L6ޔ�݄� /:���ɀ$�%|H�6�'��^o�դqE(p���o������p�(��4�qwRf�p�l5�<#y��^���N��伸l HQ&V�PY�ƃ����" -���`���TuhM� -%=D̹�]i�T�Y�n�`w���{b�{��ד`����^��� WhC��T��� ŀ�6�|>��4��|huﴯ�U-:�a���у�_'r��Uu8�p35 �sj���m�7�9G��C��G�8� ��ܘ� ޮM�!8 �]�w!t�T�r�R�h ��k��s>Xج�I��y'\\�Z�0�:lwk����I���4�����J���g`Mb�$p;���gj� ՙ�`!v�b�$1��~��'V+W=(�s������f�B -g�� !EI�#���߉k�ۘ7���"BV�|�ؽ������=�]�K试���7�)H�{��Q ���a2B�7Y�I��|�"w��]��m���w���NJb ],��p�� �gԍ�@��r�B��$3z��SD��K�qZE�r��"�B������gM+9E��Q䑽'&�'m�ν������|�|�"4�ガ�B4kY�$?8�\��G,BN����Ӣ�l^{3��ʻ1����\��=6��vn��%���q"A6�A� a�w0ۂF���A .���<(�+��ʙ� p��A���s�Q��E�RĤ2��t��Z��ܣ&CϾ��\}#5��"5�H���p-�X��'���B��4���Tm�.nG�-�;f�}������ʕ ��ɸJh���W�� ��#Z/�)����x��x+J�����*#���#p�{� -r���^�Ʈ���xJ��������EM�>�vF?�D� -��O�}wE��q��o�YtAZ�� �X{JN�� �JcZ�!��= ���ߎ�񇻨�b8 -D/�!��IK��`�� ��>�����B(H�L؃� FEO�:�� �+�LA���/ �u��:��|���ۆ�5���&� FMΏs�a���� ֎���Å,\�8m��/y�a�ݟ}�3�(��d��ןf͵����␗��uJ��\9�Q�� ^QIo�8�f�u�ڀ�G~�Ye�A>b�@�am��Kk=Vy�N�R��芫��x�܊��8����Qf^&���c�}�k+�{��%�@��b�)��t���j� j���k����P�o��Ŏ.2?t\�})C>`�D �I'��,Wّ�6?At��6�WdG��i*~��H଺���V }��p�3��d�w��<�Ҝ�o�����=��DT=�F �Mw��ح�n��ԃ54wbd�w(I�JR�N����v�� f�D��֪�V�љ�4ѧ��]c���"���/�Eg,~�m �a�f��E���@;�V]S<�-L����"����ra�E�4"�J��o���o@�te�"�0c��>�F�V� 70z �R��M�Q;���t�開�C��� [P�5���ȏ�[nx��}O��#I d�!�l�W -]���o�C�#M��ux*����ۘ��8���_)p6B3��L �mì���O1� -Iw�@�q�R�)�Ue�@V2�V��z�Dz��Ӄ>~���r$���ȸ�πW�7�ד��J;�j���d"�~�J�;�^G���9����1(<��]���ayh-�P��S�g(6�BD ��3�O7�l��a��U%���&� :>��+�� )�L�؞x�e��S�6o���}��/� �=��5������E�M�7'*j��/w�p���'E�D��"E���; ����x�m`���},o�v�J�i�=�D$�Ҽ�������NqA�?њ� h:?�����'_-+>2��[A�!č~p+F˚�Z}Ir:��Q��`��C�I���T{+��6`\gE�e��Y7��{�j�:!~��>y��b�ն��A�MH�#We�[�]Rt;%Na1��Ā 4�Uc�O�n�h0�h����VV���E\�D:J`nO��� �H��z�����X4�5,>p�d�!^'��hO����w��.K��� @� ׁ�;o"�˼"�b�j)!�EJ��7�������D댩c���j�Vn(��3@�QLn�:r�R M���a��_~��Қ7?BH{�����Q'�k� ��s��<3���o[��x�瘘�F��o�׊�rl偞�ջ p`�3uY� %?�e���J?�����`�ܛ�=��Q�W�+ۘ*��k�<�[ř;���s�0#���t�g�H����<4�3HJg�'�d���<��! ζ�z&}9B��re�ی�-`IJ�� o��S�YFo�cA�O.�L ��C�cg������G�H�p�Rp���p���,[꺂�!��I���0�LI� Zp(�^!� -��}~�KBo��t:�Zuag�������(�"�H�?z��7�2f;D�H��Bfш�%��% ��=Ϋ�օ�yT-�㼡�Nj�� -���Xk�,��@G;T|]ґ���lح�������Rȭ��P.���A\��V|~��NP�Q%�U����A銰Y�R�r��e��� ��T��1�:�=��K"���a|�&��U�Y8�?-�J�4���]��濋G?�!�pGy�����(�Vv���c �i �*��omzw%��2���5@��ܒZBx�Qy -+$�o����~����2��N�70!�=Q�"̢C�/�\?&�xj_�n�1� Ps��u��9��rȬ^�?3R5U>c�8���_�S�W5�LBϚ�1��Z�o�?��|� � �ИDW9��_lx�_/Hyb��0���;����ϫ��O���,e�ͦ��8g�G]8��;�j拱g�,uH'�rU[Y�Pͣ�E�p�.9WM�s�)ř��m�f��Χ͍��Va>D�"h<�H<*tc%���]S \ No newline at end of file diff --git a/frontends/rioterm/Cargo.toml b/frontends/rioterm/Cargo.toml index 5ad6fd20..cb5cabba 100644 --- a/frontends/rioterm/Cargo.toml +++ b/frontends/rioterm/Cargo.toml @@ -45,7 +45,6 @@ serde = { workspace = true } taffy = { version = "0.10.1", features = ["flexbox","grid"] } teletypewriter = { workspace = true } unicode-width = { workspace = true } -copa = { workspace = true } url = { workspace = true } smallvec = { workspace = true } rio-window = { workspace = true } diff --git a/rio-backend/Cargo.toml b/rio-backend/Cargo.toml index 470e2d8d..29365937 100644 --- a/rio-backend/Cargo.toml +++ b/rio-backend/Cargo.toml @@ -37,7 +37,6 @@ sugarloaf = { workspace = true } rio-grapheme-width = { workspace = true } teletypewriter = { workspace = true } unicode-width = { workspace = true } -copa = { workspace = true } # `wgpu` is only needed when the corresponding sugarloaf feature is on. wgpu = { workspace = true, optional = true } url = { workspace = true } diff --git a/rio-backend/src/ansi/sixel.rs b/rio-backend/src/ansi/sixel.rs index a40de862..78fe9e77 100644 --- a/rio-backend/src/ansi/sixel.rs +++ b/rio-backend/src/ansi/sixel.rs @@ -28,7 +28,7 @@ use std::{fmt, mem}; use crate::config::colors::ColorRgb; use sugarloaf::{ColorType, GraphicData, GraphicId, MAX_GRAPHIC_DIMENSIONS}; -use copa::Params; +use crate::performer::parser::Params; use tracing::trace; /// Type for color registers. diff --git a/rio-backend/src/batched_parser.rs b/rio-backend/src/batched_parser.rs deleted file mode 100644 index 97c5c397..00000000 --- a/rio-backend/src/batched_parser.rs +++ /dev/null @@ -1,282 +0,0 @@ -//! Enhanced Copa parser with batch UTF-8 validation -//! -//! This module provides an enhanced version of the Copa parser that uses -//! batch processing for UTF-8 validation to improve performance. - -use copa::{Parser, Perform}; - -use tracing::debug; - -/// Enhanced parser wrapper that adds batch UTF-8 processing -pub struct BatchedParser { - /// The underlying Copa parser - parser: Parser, - /// Buffer for accumulating input chunks - input_buffer: Vec, - /// Threshold for triggering batch processing - batch_threshold: usize, - /// Performance statistics - stats: BatchStats, -} - -impl Default for BatchedParser { - fn default() -> Self { - Self::new() - } -} - -impl BatchedParser { - /// Create a new batched parser with optimal defaults - pub fn new() -> Self { - Self { - parser: Parser::::default(), - input_buffer: Vec::with_capacity(4096), - batch_threshold: 1024, // 1KB threshold - optimal for terminal usage - stats: BatchStats::default(), - } - } - - /// Process input with potential batching - pub fn advance(&mut self, performer: &mut P, bytes: &[u8]) { - // Only batch for very large inputs (paste operations, large TUI output) - // Process normal terminal input immediately for responsiveness - if bytes.len() < self.batch_threshold { - self.stats.record_immediate(bytes.len()); - - debug!("BatchedParser: immediate processing {} bytes", bytes.len()); - - self.parser.advance(performer, bytes); - return; - } - - // For large inputs, use batching - self.input_buffer.extend_from_slice(bytes); - - debug!( - "BatchedParser: added {} bytes to buffer (total: {})", - bytes.len(), - self.input_buffer.len() - ); - - // Process immediately if we have a large batch - if self.input_buffer.len() >= self.batch_threshold { - let batch_size = self.input_buffer.len(); - self.stats.record_batch(batch_size); - - debug!("BatchedParser: flushing batch of {} bytes", batch_size); - - self.flush_batch(performer); - } - } - - /// Force flush any pending batched input - pub fn flush(&mut self, performer: &mut P) { - if !self.input_buffer.is_empty() { - self.flush_batch(performer); - } - } - - /// Internal method to flush the current batch - fn flush_batch(&mut self, performer: &mut P) { - if self.input_buffer.is_empty() { - return; - } - - // Process the entire buffer at once - self.parser.advance(performer, &self.input_buffer); - - // Clear the buffer and shrink if it's grown too large - self.input_buffer.clear(); - - // Prevent memory bloat by shrinking oversized buffers - if self.input_buffer.capacity() > 16384 { - self.input_buffer.shrink_to(4096); - } - } - - /// Get the underlying parser (for compatibility) - pub fn inner(&self) -> &Parser { - &self.parser - } - - /// Get mutable access to the underlying parser - pub fn inner_mut(&mut self) -> &mut Parser { - &mut self.parser - } - - /// Get current buffer size for monitoring - pub fn buffer_len(&self) -> usize { - self.input_buffer.len() - } - - /// Get performance statistics - pub fn stats(&self) -> &BatchStats { - &self.stats - } - - /// Reset performance statistics - pub fn reset_stats(&mut self) { - self.stats = BatchStats::default(); - } - - /// Get current batch threshold - pub fn batch_threshold(&self) -> usize { - self.batch_threshold - } - - /// Process input until terminated, compatible with Copa parser interface - pub fn advance_until_terminated( - &mut self, - performer: &mut P, - bytes: &[u8], - ) -> usize { - // Only batch for very large inputs (paste operations, large TUI output) - // Process normal terminal input immediately for responsiveness - if bytes.len() < self.batch_threshold { - self.stats.record_immediate(bytes.len()); - return self.parser.advance_until_terminated(performer, bytes); - } - - // For large inputs, use batching - self.input_buffer.extend_from_slice(bytes); - let bytes_added = bytes.len(); - - // Process immediately if we have a large batch - if self.input_buffer.len() >= self.batch_threshold { - let batch_size = self.input_buffer.len(); - self.stats.record_batch(batch_size); - self.flush_batch(performer); - } - - // Always return the number of bytes we just processed - bytes_added - } -} - -/// Statistics for batch processing performance monitoring -#[derive(Debug, Default)] -pub struct BatchStats { - /// Total bytes processed - pub total_bytes: usize, - /// Number of batch operations - pub batch_count: usize, - /// Number of immediate (non-batched) operations - pub immediate_count: usize, - /// Average batch size - pub avg_batch_size: f64, -} - -impl BatchStats { - /// Update stats with a new batch - pub fn record_batch(&mut self, batch_size: usize) { - self.total_bytes += batch_size; - self.batch_count += 1; - self.update_average(); - } - - /// Update stats with an immediate operation - pub fn record_immediate(&mut self, size: usize) { - self.total_bytes += size; - self.immediate_count += 1; - } - - /// Update the average batch size - fn update_average(&mut self) { - if self.batch_count > 0 { - self.avg_batch_size = self.total_bytes as f64 / self.batch_count as f64; - } - } - - /// Get the batching efficiency (percentage of bytes processed in batches) - pub fn batch_efficiency(&self) -> f64 { - if self.total_bytes == 0 { - return 0.0; - } - - let batched_bytes = self.batch_count as f64 * self.avg_batch_size; - (batched_bytes / self.total_bytes as f64) * 100.0 - } -} - -#[cfg(test)] -mod tests { - use super::*; - use copa::Perform; - - // Mock performer for testing - struct MockPerformer { - chars_received: Vec, - } - - impl Perform for MockPerformer { - fn print(&mut self, c: char) { - self.chars_received.push(c); - } - - fn execute(&mut self, _byte: u8) {} - fn hook( - &mut self, - _params: &copa::Params, - _intermediates: &[u8], - _ignore: bool, - _c: char, - ) { - } - fn put(&mut self, _byte: u8) {} - fn unhook(&mut self) {} - fn osc_dispatch(&mut self, _params: &[&[u8]], _bell_terminated: bool) {} - fn csi_dispatch( - &mut self, - _params: &copa::Params, - _intermediates: &[u8], - _ignore: bool, - _c: char, - ) { - } - fn esc_dispatch(&mut self, _intermediates: &[u8], _ignore: bool, _byte: u8) {} - } - - #[test] - fn test_batched_parser_small_input() { - let mut parser = BatchedParser::<1024>::new(); - let mut performer = MockPerformer { - chars_received: Vec::new(), - }; - - // Small input should be processed immediately - parser.advance(&mut performer, b"Hello"); - - // Should have processed the text - assert_eq!(performer.chars_received, vec!['H', 'e', 'l', 'l', 'o']); - } - - #[test] - fn test_batched_parser_large_input() { - let mut parser = BatchedParser::<1024>::new(); - let mut performer = MockPerformer { - chars_received: Vec::new(), - }; - - // Large input that should trigger batching - let large_input = "A".repeat(1000); - parser.advance(&mut performer, large_input.as_bytes()); - - // Should have processed all characters - assert_eq!(performer.chars_received.len(), 1000); - assert!(performer.chars_received.iter().all(|&c| c == 'A')); - } - - #[test] - fn test_batch_stats() { - let mut stats = BatchStats::default(); - - stats.record_batch(100); - stats.record_batch(200); - stats.record_immediate(50); - - assert_eq!(stats.total_bytes, 350); - assert_eq!(stats.batch_count, 2); - assert_eq!(stats.immediate_count, 1); - assert_eq!(stats.avg_batch_size, 150.0); - } -} diff --git a/rio-backend/src/crosswords/grid/row.rs b/rio-backend/src/crosswords/grid/row.rs index 580ea37c..d6cbcd75 100644 --- a/rio-backend/src/crosswords/grid/row.rs +++ b/rio-backend/src/crosswords/grid/row.rs @@ -22,10 +22,7 @@ pub struct Row { /// Set when at least one cell in the row contains a kitty Unicode /// graphics-protocol placeholder (U+10EEEE). The renderer skips the - /// placeholder scan on rows where this is `false`. Mirrors ghostty's - /// `page.zig:1953-1958` `kitty_virtual_placeholder` row flag. - /// Cleared by `reset`; set by `Crosswords::input` whenever a - /// placeholder codepoint lands in the row. + /// placeholder scan on rows where this is `false`. pub kitty_virtual_placeholder: bool, } diff --git a/rio-backend/src/crosswords/mod.rs b/rio-backend/src/crosswords/mod.rs index ea25f4fd..6b963d14 100644 --- a/rio-backend/src/crosswords/mod.rs +++ b/rio-backend/src/crosswords/mod.rs @@ -43,12 +43,12 @@ use crate::crosswords::square::{CellFlags, Wide}; use crate::event::WindowId; use crate::event::{EventListener, RioEvent, TerminalDamage}; use crate::performer::handler::Handler; +use crate::performer::parser::Params; use crate::selection::{Selection, SelectionRange, SelectionType}; use crate::simd_utf8; use attr::*; use base64::{engine::general_purpose, Engine as _}; use bitflags::bitflags; -use copa::Params; use grid::row::Row; use pos::{ Boundary, CharsetIndex, Column, Cursor, CursorState, Direction, Line, Pos, Side, diff --git a/rio-backend/src/graphics/kitty/mod.rs b/rio-backend/src/graphics/kitty/mod.rs index a372e1ec..1b6b022a 100644 --- a/rio-backend/src/graphics/kitty/mod.rs +++ b/rio-backend/src/graphics/kitty/mod.rs @@ -272,10 +272,9 @@ fn test_chunked_transfer() { let mut state = KittyGraphicsState::default(); // Total base64 for 1x1 RGBA pixel [255, 0, 0, 255] is "/wAA/w==". - // Each chunk is decoded independently now (matching ghostty / chafa - // style), so each must be a valid base64 on its own — either a - // multiple of 4 chars per kitty spec, or an independently padded - // chunk. Here we use two spec-compliant chunks. + // Each chunk is decoded independently now, so each must be a + // valid base64 on its own — either a multiple of 4 chars per kitty + // spec, or an independently padded chunk. // Chunk 1 (m=1): 4 chars → 3 decoded bytes [0xFF, 0x00, 0x00] let params1 = vec![ diff --git a/rio-backend/src/lib.rs b/rio-backend/src/lib.rs index 273368a5..1037db97 100644 --- a/rio-backend/src/lib.rs +++ b/rio-backend/src/lib.rs @@ -1,6 +1,5 @@ pub mod ansi; pub mod batch_utf8; -pub mod batched_parser; pub mod clipboard; pub mod config; pub mod crosswords; diff --git a/rio-backend/src/performer/handler.rs b/rio-backend/src/performer/handler.rs index 49d31366..2c243b16 100644 --- a/rio-backend/src/performer/handler.rs +++ b/rio-backend/src/performer/handler.rs @@ -3,7 +3,6 @@ use crate::ansi::iterm2_image_protocol; use crate::ansi::kitty_graphics_protocol; use crate::ansi::CursorShape; use crate::ansi::{sixel, KeyboardModes, KeyboardModesApplyBehavior}; -use crate::batched_parser::BatchedParser; use crate::config::colors::{AnsiColor, ColorRgb, NamedColor}; use crate::crosswords::pos::{CharsetIndex, Column, Line, StandardCharset}; use crate::crosswords::square::Hyperlink; @@ -26,7 +25,7 @@ use crate::ansi::{ use std::fmt::Write; // https://vt100.net/emu/dec_ansi_parser -use copa::{Params, ParamsIter}; +use super::parser::{Params, ParamsIter, Parser, Perform}; /// Maximum time before a synchronized update is aborted. const SYNC_UPDATE_TIMEOUT: Duration = Duration::from_millis(150); @@ -604,7 +603,7 @@ impl Timeout for StdSyncHandler { #[derive(Default)] pub struct Processor { state: ProcessorState, - parser: BatchedParser<1024>, + parser: Parser, } impl Processor { @@ -624,29 +623,14 @@ impl Processor { where H: Handler, { - let mut processed = 0; - while processed != bytes.len() { - if self.state.sync_state.timeout.pending_timeout() { - processed += self.advance_sync(handler, &bytes[processed..]); - } else { - let mut performer = Performer::new(&mut self.state, handler); - processed += self - .parser - .advance_until_terminated(&mut performer, &bytes[processed..]); - } + if self.state.sync_state.timeout.pending_timeout() { + self.advance_sync(handler, bytes); + } else { + let mut performer = Performer::new(&mut self.state, handler); + self.parser.advance(&mut performer, bytes); } } - /// Flush any pending batched input - #[inline] - pub fn flush(&mut self, handler: &mut H) - where - H: Handler, - { - let mut performer = Performer::new(&mut self.state, handler); - self.parser.flush(&mut performer); - } - /// End a synchronized update. pub fn stop_sync(&mut self, handler: &mut H) where @@ -664,16 +648,12 @@ impl Processor { where H: Handler, { - // Process all synchronized bytes. - // - // NOTE: We do not use `advance_until_terminated` here since BSU sequences are - // processed automatically during the synchronized update. + // Process all synchronized bytes. BSU sequences are processed + // automatically during the synchronized update. let buffer = mem::take(&mut self.state.sync_state.buffer); let offset = bsu_offset.unwrap_or(buffer.len()); let mut performer = Performer::new(&mut self.state, handler); self.parser.advance(&mut performer, &buffer[..offset]); - // Flush any pending batched input from synchronized processing - self.parser.flush(&mut performer); self.state.sync_state.buffer = buffer; match bsu_offset { @@ -702,10 +682,8 @@ impl Processor { } /// Process a new byte during a synchronized update. - /// - /// Returns the number of bytes processed. #[cold] - fn advance_sync(&mut self, handler: &mut H, bytes: &[u8]) -> usize + fn advance_sync(&mut self, handler: &mut H, bytes: &[u8]) where H: Handler, { @@ -716,11 +694,10 @@ impl Processor { // Just parse the bytes normally. let mut performer = Performer::new(&mut self.state, handler); - self.parser.advance_until_terminated(&mut performer, bytes) + self.parser.advance(&mut performer, bytes); } else { self.state.sync_state.buffer.extend(bytes); self.advance_sync_csi(handler, bytes.len()); - bytes.len() } } @@ -982,7 +959,7 @@ impl<'a, H: Handler + 'a, T: Timeout> Performer<'a, H, T> { } } -impl copa::Perform for Performer<'_, U, T> { +impl Perform for Performer<'_, U, T> { fn print(&mut self, c: char) { self.handler.input(c); self.state.preceding_char = Some(c); @@ -1668,10 +1645,7 @@ impl copa::Perform for Performer<'_, U, T> { self.state.apc_state.buffer.push(byte); } - /// Called when the APC sequence ends. - /// - /// Process the accumulated APC data. This is called instead of apc_dispatch - /// when we implement custom APC hooks. + /// Called when the APC sequence ends. Processes the accumulated APC data. fn apc_end(&mut self) { debug!( "[apc_end] APC complete, accumulated {} bytes", diff --git a/rio-backend/src/performer/mod.rs b/rio-backend/src/performer/mod.rs index 3dee2715..60f82f14 100644 --- a/rio-backend/src/performer/mod.rs +++ b/rio-backend/src/performer/mod.rs @@ -1,4 +1,5 @@ pub mod handler; +pub mod parser; use crate::crosswords::Crosswords; use crate::event::sync::FairMutex; diff --git a/copa/src/lib.rs b/rio-backend/src/performer/parser/mod.rs similarity index 80% rename from copa/src/lib.rs rename to rio-backend/src/performer/parser/mod.rs index 1fd62d62..180d9f6c 100644 --- a/copa/src/lib.rs +++ b/rio-backend/src/performer/parser/mod.rs @@ -1,43 +1,20 @@ -//! Parser for implementing virtual terminal emulators +//! Parser for virtual terminal escape sequences. //! -//! [`Parser`] is implemented according to [Paul Williams' ANSI parser -//! state machine]. The state machine doesn't assign meaning to the parsed data -//! and is thus not itself sufficient for writing a terminal emulator. Instead, -//! it is expected that an implementation of [`Perform`] is provided which does -//! something useful with the parsed data. The [`Parser`] handles the book -//! keeping, and the [`Perform`] gets to simply handle actions. +//! [`Parser`] implements [Paul Williams' ANSI parser state machine]. The state +//! machine doesn't assign meaning to the parsed data — that's the job of the +//! [`Perform`] implementer. //! -//! # Examples +//! Forked from Alacritty's VTE; previously the standalone `copa` crate. The +//! crate-private [`Perform`] trait keeps a single dispatch shape so the same +//! state machine drives both the production [`Performer`] and unit-test +//! dispatchers. //! -//! For an example of using the [`Parser`] please see the examples folder. The example included -//! there simply logs all the actions [`Perform`] does. One quick thing to see it in action is to -//! pipe `vim` into it -//! -//! ```sh -//! cargo build --release --example parselog -//! vim | target/release/examples/parselog -//! ``` -//! -//! Just type `:q` to exit. -//! -//! # Differences from original state machine description -//! -//! * UTF-8 Support for Input -//! * OSC Strings can be terminated by 0x07 -//! * Only supports 7-bit codes. Some 8-bit codes are still supported, but they no longer work in -//! all states. -//! -//! [`Parser`]: struct.Parser.html -//! [`Perform`]: trait.Perform.html //! [Paul Williams' ANSI parser state machine]: https://vt100.net/emu/dec_ansi_parser -#![deny(clippy::all, clippy::if_not_else, clippy::enum_glob_use)] -#![cfg_attr(not(feature = "std"), no_std)] +//! [`Performer`]: super::handler::Performer -use core::mem::MaybeUninit; -use core::str; +#![deny(clippy::all, clippy::if_not_else, clippy::enum_glob_use)] -#[cfg(not(feature = "std"))] -use arrayvec::ArrayVec; +use std::str; mod params; @@ -45,25 +22,22 @@ pub use params::{Params, ParamsIter}; const MAX_INTERMEDIATES: usize = 2; const MAX_OSC_PARAMS: usize = 16; -const MAX_OSC_RAW: usize = 1024; -/// Parser for raw _VTE_ protocol which delegates actions to a [`Perform`] -/// -/// [`Perform`]: trait.Perform.html -/// -/// Generic over the value for the size of the raw Operating System Command -/// buffer. Only used when the `std` feature is not enabled. +/// Inline OSC byte capacity. Sized to absorb common OSCs (titles, color +/// queries, hyperlink URLs, kitty graphics control headers) without +/// allocation. Larger payloads (e.g. OSC 52 clipboard pastes) spill into +/// `OscBuffer::overflow`. +const OSC_FIXED_LEN: usize = 2048; + +/// Parser for raw _VTE_ protocol which delegates actions to a [`Perform`]. #[derive(Default)] -pub struct Parser { +pub(crate) struct Parser { state: State, intermediates: [u8; MAX_INTERMEDIATES], intermediate_idx: usize, params: Params, param: u16, - #[cfg(not(feature = "std"))] - osc_raw: ArrayVec, - #[cfg(feature = "std")] - osc_raw: Vec, + osc_raw: OscBuffer, osc_params: [(usize, usize); MAX_OSC_PARAMS], osc_num_params: usize, ignoring: bool, @@ -71,27 +45,74 @@ pub struct Parser { partial_utf8_len: usize, } -impl Parser { - /// Create a new Parser - pub fn new() -> Parser { - Default::default() +/// OSC accumulator with a fixed-size inline buffer and a heap fallback. +/// +/// The first `OSC_FIXED_LEN` bytes of any OSC sequence land in `fixed` +/// (zero allocation). On overflow, the populated prefix of `fixed` is copied +/// into `overflow` once and all subsequent writes go to the `Vec` only — so +/// at any moment a single backing slice holds the contiguous payload. +struct OscBuffer { + fixed: [u8; OSC_FIXED_LEN], + fixed_len: usize, + overflow: Vec, +} + +impl Default for OscBuffer { + fn default() -> Self { + Self { + fixed: [0; OSC_FIXED_LEN], + fixed_len: 0, + overflow: Vec::new(), + } } } -impl Parser { - /// Create a new Parser with a custom size for the Operating System Command - /// buffer. - /// - /// Call with a const-generic param on `Parser`, like: - /// - /// ```rust - /// let mut p = copa::Parser::<64>::new_with_size(); - /// ``` - #[cfg(not(feature = "std"))] - pub fn new_with_size() -> Parser { - Default::default() +impl OscBuffer { + #[inline] + fn len(&self) -> usize { + if self.overflow.is_empty() { + self.fixed_len + } else { + self.overflow.len() + } + } + + #[inline] + fn push(&mut self, byte: u8) { + if self.overflow.is_empty() { + if self.fixed_len < OSC_FIXED_LEN { + self.fixed[self.fixed_len] = byte; + self.fixed_len += 1; + return; + } + // Spill: promote the current contents to the heap once, then + // append. After this point, `overflow.len() >= OSC_FIXED_LEN`, + // so the `is_empty()` check above stays false until `clear`. + self.overflow + .extend_from_slice(&self.fixed[..self.fixed_len]); + } + self.overflow.push(byte); } + #[inline] + fn slice(&self, start: usize, end: usize) -> &[u8] { + if self.overflow.is_empty() { + &self.fixed[start..end] + } else { + &self.overflow[start..end] + } + } + + #[inline] + fn clear(&mut self) { + self.fixed_len = 0; + // Keep `overflow`'s capacity so a session that hits one large paste + // doesn't re-allocate on the next one. + self.overflow.clear(); + } +} + +impl Parser { #[inline] fn params(&self) -> &Params { &self.params @@ -105,10 +126,8 @@ impl Parser { /// Advance the parser state. /// /// Requires a [`Perform`] implementation to handle the triggered actions. - /// - /// [`Perform`]: trait.Perform.html #[inline] - pub fn advance(&mut self, performer: &mut P, bytes: &[u8]) { + pub(crate) fn advance(&mut self, performer: &mut P, bytes: &[u8]) { let mut i = 0; // Handle partial codepoints from previous calls to `advance`. @@ -129,43 +148,6 @@ impl Parser { } } - /// Partially advance the parser state. - /// - /// This is equivalent to [`Self::advance`], but stops when - /// [`Perform::terminated`] is true after reading a byte. - /// - /// Returns the number of bytes read before termination. - /// - /// See [`Perform::advance`] for more details. - #[inline] - #[must_use = "Returned value should be used to processs the remaining bytes"] - pub fn advance_until_terminated( - &mut self, - performer: &mut P, - bytes: &[u8], - ) -> usize { - let mut i = 0; - - // Handle partial codepoints from previous calls to `advance`. - if self.partial_utf8_len != 0 { - i += self.advance_partial_utf8(performer, bytes); - } - - while i != bytes.len() && !performer.terminated() { - match self.state { - State::Ground => i += self.advance_ground(performer, &bytes[i..]), - _ => { - // Inlining it results in worse codegen. - let byte = bytes[i]; - self.change_state(performer, byte); - i += 1; - } - } - } - - i - } - #[inline(always)] fn change_state(&mut self, performer: &mut P, byte: u8) { match self.state { @@ -181,9 +163,9 @@ impl Parser { State::Escape => self.advance_esc(performer, byte), State::EscapeIntermediate => self.advance_esc_intermediate(performer, byte), State::OscString => self.advance_osc_string(performer, byte), - State::SosString => self.advance_opaque_string(SosDispatch(performer), byte), + State::SosString => self.advance_sos_string(performer, byte), State::ApcString => self.advance_apc_string(performer, byte), - State::PmString => self.advance_opaque_string(PmDispatch(performer), byte), + State::PmString => self.advance_pm_string(performer, byte), State::Ground => unreachable!(), } } @@ -435,105 +417,78 @@ impl Parser { self.reset_params(); self.state = State::Escape } - 0x3B => { - #[cfg(not(feature = "std"))] - { - if self.osc_raw.is_full() { - return; - } - } - self.action_osc_put_param() - } + 0x3B => self.action_osc_put_param(), _ => self.action_osc_put(byte), } } #[inline(always)] fn advance_apc_string(&mut self, performer: &mut P, byte: u8) { + // Bytes stream straight through to `performer.apc_put`; the Performer + // owns its own accumulation buffer (`apc_state.buffer`) and parses + // kitty-style headers from there. The parser keeps no APC state. match byte { 0x00..=0x06 | 0x08..=0x17 | 0x19 | 0x1C..=0x1F => (), // Ignore control bytes 0x07 => { - // Bell-terminated APC - self.action_apc_end(performer); - self.osc_raw.clear(); - self.osc_num_params = 0; + // Bell-terminated APC. + performer.apc_end(); self.state = State::Ground; } 0x18 | 0x1A => { - // C0 termination (CAN or SUB) - self.action_apc_put(performer, byte); - self.action_apc_end(performer); + // C0 termination (CAN or SUB). + performer.apc_put(byte); + performer.apc_end(); performer.execute(byte); - self.osc_raw.clear(); - self.osc_num_params = 0; self.state = State::Ground; } 0x1B => { - // Start of ST termination (\x1b\) - self.action_apc_end(performer); - self.osc_raw.clear(); - self.osc_num_params = 0; + // Start of ST termination (`\x1b\`). + performer.apc_end(); self.state = State::Escape; } - 0x3B => { - // Semicolon separates control data from payload - #[cfg(not(feature = "std"))] - { - if self.osc_raw.is_full() { - return; - } - } - self.action_apc_put(performer, byte); - self.action_osc_put_param(); // Reuse existing method to track parameter boundaries - } - 0x2C => { - // Comma is part of the control data (separates key-value pairs) - // Don't create a parameter boundary - self.action_apc_put(performer, byte); - } - 0x20..=0xFF => { - // Collect valid APC content (control data or payload) - self.action_apc_put(performer, byte); - } + 0x20..=0xFF => performer.apc_put(byte), } } #[inline(always)] - fn action_apc_put(&mut self, performer: &mut P, byte: u8) { - #[cfg(not(feature = "std"))] - { - if self.osc_raw.is_full() { - return; + fn advance_sos_string(&mut self, performer: &mut P, byte: u8) { + match byte { + 0x07 => { + performer.sos_end(); + self.state = State::Ground } + 0x18 | 0x1A => { + performer.sos_end(); + performer.execute(byte); + self.state = State::Ground + } + 0x1B => { + performer.sos_end(); + self.state = State::Escape + } + 0x20..=0xFF => performer.sos_put(byte), + // Ignore all other control bytes. + _ => (), } - self.osc_raw.push(byte); - performer.apc_put(byte); // Optionally pass to apc_put for immediate processing - } - - #[inline] - fn action_apc_end(&self, performer: &mut P) { - // APCs are handled through apc_start/apc_put/apc_end hooks which properly - // accumulate large payloads. This function just calls apc_end() to complete the sequence. - performer.apc_end(); } #[inline(always)] - fn advance_opaque_string(&mut self, mut dispatcher: D, byte: u8) { + fn advance_pm_string(&mut self, performer: &mut P, byte: u8) { match byte { 0x07 => { - dispatcher.opaque_end(); + performer.pm_end(); self.state = State::Ground } 0x18 | 0x1A => { - dispatcher.opaque_end(); - dispatcher.execute(byte); + performer.pm_end(); + performer.execute(byte); self.state = State::Ground } 0x1B => { - dispatcher.opaque_end(); + performer.pm_end(); self.state = State::Escape } - 0x20..=0xFF => dispatcher.opaque_put(byte), + 0x20..=0xFF => performer.pm_put(byte), // Ignore all other control bytes. _ => (), } @@ -657,12 +612,6 @@ impl Parser { #[inline(always)] fn action_osc_put(&mut self, byte: u8) { - #[cfg(not(feature = "std"))] - { - if self.osc_raw.is_full() { - return; - } - } self.osc_raw.push(byte); } @@ -683,25 +632,20 @@ impl Parser { self.params.clear(); } - /// Separate method for osc_dispatch that borrows self as read-only + /// Separate method for osc_dispatch that borrows self as read-only. /// - /// The aliasing is needed here for multiple slices into self.osc_raw + /// The aliasing is needed here for multiple slices into self.osc_raw. #[inline] fn osc_dispatch(&self, performer: &mut P, byte: u8) { - let mut slices: [MaybeUninit<&[u8]>; MAX_OSC_PARAMS] = - unsafe { MaybeUninit::uninit().assume_init() }; - - for (i, slice) in slices.iter_mut().enumerate().take(self.osc_num_params) { - let indices = self.osc_params[i]; - *slice = MaybeUninit::new(&self.osc_raw[indices.0..indices.1]); - } - - unsafe { - let num_params = self.osc_num_params; - let params = &slices[..num_params] as *const [MaybeUninit<&[u8]>] - as *const [&[u8]]; - performer.osc_dispatch(&*params, byte == 0x07); + let mut slices: [&[u8]; MAX_OSC_PARAMS] = [&[]; MAX_OSC_PARAMS]; + for (slice, &(start, end)) in slices + .iter_mut() + .zip(self.osc_params.iter()) + .take(self.osc_num_params) + { + *slice = self.osc_raw.slice(start, end); } + performer.osc_dispatch(&slices[..self.osc_num_params], byte == 0x07); } /// Advance the parser state from ground. @@ -805,6 +749,9 @@ impl Parser { match simdutf8::basic::from_utf8(&self.partial_utf8[..self.partial_utf8_len]) { // If the entire buffer is valid, use the first character and continue parsing. Ok(parsed) => { + // SAFETY: `partial_utf8_len >= 1` (caller guarantee) and `parsed` + // is the validated UTF-8 view of those bytes, so it has at least + // one character. let c = unsafe { parsed.chars().next().unwrap_unchecked() }; performer.print(c); @@ -823,6 +770,9 @@ impl Parser { // utf8 character into `partial_utf8`. Since we only care about the // first character, we just ignore the rest. if valid_bytes > 0 { + // SAFETY: `valid_bytes > 0` and the slice up to `valid_bytes` was + // reported as valid UTF-8 by the compat decoder, so it contains + // at least one full character. let c = unsafe { let parsed = str::from_utf8_unchecked(&self.partial_utf8[..valid_bytes]); @@ -884,17 +834,15 @@ enum State { Ground, } -/// Performs actions requested by the Parser +/// Performs actions requested by the [`Parser`]. /// -/// Actions in this case mean, for example, handling a CSI escape sequence -/// describing cursor movement, or simply printing characters to the screen. +/// Crate-private dispatch trait. The single production implementer is +/// [`super::handler::Performer`]; tests in this module supply their own +/// recording dispatchers. /// -/// The methods on this type correspond to actions described in -/// . I've done my best to describe them in -/// a useful way in my own words for completeness, but the site should be -/// referenced if something isn't clear. If the site disappears at some point in -/// the future, consider checking archive.org. -pub trait Perform { +/// The methods correspond to actions described in +/// . +pub(crate) trait Perform { /// Draw a character to the screen and update states. fn print(&mut self, _c: char) {} @@ -934,9 +882,6 @@ pub trait Perform { /// Dispatch an operating system command. fn osc_dispatch(&mut self, _params: &[&[u8]], _bell_terminated: bool) {} - /// Dispatch an application program command. - fn apc_dispatch(&mut self, _params: &[&[u8]], _bell_terminated: bool) {} - /// A final character has arrived for a CSI sequence /// /// The `ignore` flag indicates that either more than two intermediates @@ -990,97 +935,10 @@ pub trait Perform { /// Invoked when the end of an APC (Application Program Command) sequence is /// encountered. fn apc_end(&mut self) {} - - /// Whether the parser should terminate prematurely. - /// - /// This can be used in conjunction with - /// [`Parser::advance_until_terminated`] to terminate the parser after - /// receiving certain escape sequences like synchronized updates. - /// - /// This is checked after every parsed byte, so no expensive computation - /// should take place in this function. - #[inline(always)] - fn terminated(&self) -> bool { - false - } } -/// This trait is used internally to provide a common implementation for Opaque -/// Sequences (SOS, APC, PM). Implementations of this trait will just forward -/// calls to the equivalent method on [Perform]. Implementations of this trait -/// are always inlined to avoid overhead. -trait OpaqueDispatch { - fn execute(&mut self, byte: u8); - fn opaque_put(&mut self, byte: u8); - fn opaque_end(&mut self); -} - -struct SosDispatch<'a, P: Perform>(&'a mut P); - -impl OpaqueDispatch for SosDispatch<'_, P> { - #[inline(always)] - fn execute(&mut self, byte: u8) { - self.0.execute(byte); - } - - #[inline(always)] - fn opaque_put(&mut self, byte: u8) { - self.0.sos_put(byte); - } - - #[inline(always)] - fn opaque_end(&mut self) { - self.0.sos_end(); - } -} - -#[allow(dead_code)] -struct ApcDispatch<'a, P: Perform>(&'a mut P); - -impl OpaqueDispatch for ApcDispatch<'_, P> { - #[inline(always)] - fn execute(&mut self, byte: u8) { - self.0.execute(byte); - } - - #[inline(always)] - fn opaque_put(&mut self, byte: u8) { - self.0.apc_put(byte); - } - - #[inline(always)] - fn opaque_end(&mut self) { - self.0.apc_end(); - } -} - -struct PmDispatch<'a, P: Perform>(&'a mut P); - -impl OpaqueDispatch for PmDispatch<'_, P> { - #[inline(always)] - fn execute(&mut self, byte: u8) { - self.0.execute(byte); - } - - #[inline(always)] - fn opaque_put(&mut self, byte: u8) { - self.0.pm_put(byte); - } - - #[inline(always)] - fn opaque_end(&mut self) { - self.0.pm_end(); - } -} - -#[cfg(all(test, not(feature = "std")))] -#[macro_use] -extern crate std; - #[cfg(test)] mod tests { - use std::vec::Vec; - use super::*; const OSC_BYTES: &[u8] = &[ @@ -1216,7 +1074,7 @@ mod tests { #[test] fn parse_osc() { let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, OSC_BYTES); @@ -1234,7 +1092,7 @@ mod tests { #[test] fn parse_empty_osc() { let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, &[0x1B, 0x5D, 0x07]); @@ -1250,7 +1108,7 @@ mod tests { let params = ";".repeat(params::MAX_PARAMS + 1); let input = format!("\x1b]{}\x1b", ¶ms[..]).into_bytes(); let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, &input); @@ -1268,7 +1126,7 @@ mod tests { fn osc_bell_terminated() { const INPUT: &[u8] = b"\x1b]11;ff/00/ff\x07"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1283,7 +1141,7 @@ mod tests { fn osc_c0_st_terminated() { const INPUT: &[u8] = b"\x1b]11;ff/00/ff\x1b\\"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1302,7 +1160,7 @@ mod tests { 0x26, 0x26, 0x20, 0x73, 0x6C, 0x65, 0x65, 0x70, 0x20, 0x31, 0x07, ]; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1319,7 +1177,7 @@ mod tests { fn osc_containing_string_terminator() { const INPUT: &[u8] = b"\x1b]2;\xe6\x9c\xab\x1b\\"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1333,34 +1191,53 @@ mod tests { } #[test] - fn exceed_max_buffer_size() { - const NUM_BYTES: usize = MAX_OSC_RAW + 100; + fn osc_fits_in_inline_buffer() { + // Stay below `OSC_FIXED_LEN`; the spill `Vec` should never grow. + const NUM_BYTES: usize = OSC_FIXED_LEN - 32; const INPUT_START: &[u8] = b"\x1b]52;s"; const INPUT_END: &[u8] = b"\x07"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); - // Create valid OSC escape parser.advance(&mut dispatcher, INPUT_START); - - // Exceed max buffer size parser.advance(&mut dispatcher, &[b'a'; NUM_BYTES]); - - // Terminate escape for dispatch parser.advance(&mut dispatcher, INPUT_END); + assert!(parser.osc_raw.overflow.capacity() == 0); assert_eq!(dispatcher.dispatched.len(), 1); match &dispatcher.dispatched[0] { Sequence::Osc(params, _) => { assert_eq!(params.len(), 2); assert_eq!(params[0], b"52"); - - #[cfg(feature = "std")] assert_eq!(params[1].len(), NUM_BYTES + INPUT_END.len()); + } + _ => panic!("expected osc sequence"), + } + } + + #[test] + fn osc_spills_to_overflow() { + // Push past `OSC_FIXED_LEN` to exercise the heap-fallback path. + const NUM_BYTES: usize = OSC_FIXED_LEN + 512; + const INPUT_START: &[u8] = b"\x1b]52;s"; + const INPUT_END: &[u8] = b"\x07"; + + let mut dispatcher = Dispatcher::default(); + let mut parser = Parser::default(); - #[cfg(not(feature = "std"))] - assert_eq!(params[1].len(), MAX_OSC_RAW - params[0].len()); + parser.advance(&mut dispatcher, INPUT_START); + parser.advance(&mut dispatcher, &[b'a'; NUM_BYTES]); + parser.advance(&mut dispatcher, INPUT_END); + + assert_eq!(dispatcher.dispatched.len(), 1); + match &dispatcher.dispatched[0] { + Sequence::Osc(params, _) => { + assert_eq!(params.len(), 2); + assert_eq!(params[0], b"52"); + assert_eq!(params[1].len(), NUM_BYTES + INPUT_END.len()); + assert_eq!(params[1][0], b's'); + assert!(params[1][1..].iter().all(|&b| b == b'a')); } _ => panic!("expected osc sequence"), } @@ -1375,7 +1252,7 @@ mod tests { let input = format!("\x1b[{}p", ¶ms[..]).into_bytes(); let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, &input); @@ -1398,7 +1275,7 @@ mod tests { let input = format!("\x1b[{}p", ¶ms[..]).into_bytes(); let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, &input); @@ -1415,7 +1292,7 @@ mod tests { #[test] fn parse_csi_params_trailing_semicolon() { let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, b"\x1b[4;m"); @@ -1430,7 +1307,7 @@ mod tests { fn parse_csi_params_leading_semicolon() { // Create dispatcher and check state let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, b"\x1b[;4m"); @@ -1446,7 +1323,7 @@ mod tests { // The important part is the parameter, which is (i64::MAX + 1) const INPUT: &[u8] = b"\x1b[9223372036854775808m"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1461,7 +1338,7 @@ mod tests { fn csi_reset() { const INPUT: &[u8] = b"\x1b[3;1\x1b[?1049h"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1480,7 +1357,7 @@ mod tests { fn csi_subparameters() { const INPUT: &[u8] = b"\x1b[38:2:255:0:255;1m"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1500,7 +1377,7 @@ mod tests { let params = "1;".repeat(params::MAX_PARAMS + 1); let input = format!("\x1bP{}p", ¶ms[..]).into_bytes(); let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, &input); @@ -1519,7 +1396,7 @@ mod tests { fn dcs_reset() { const INPUT: &[u8] = b"\x1b[3;1\x1bP1$tx\x9c"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1540,7 +1417,7 @@ mod tests { fn parse_dcs() { const INPUT: &[u8] = b"\x1bP0;1|17/ab\x9c"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1562,7 +1439,7 @@ mod tests { fn intermediate_reset_on_dcs_exit() { const INPUT: &[u8] = b"\x1bP=1sZZZ\x1b+\x5c"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1577,7 +1454,7 @@ mod tests { fn esc_reset() { const INPUT: &[u8] = b"\x1b[3;1\x1b(A"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1596,7 +1473,7 @@ mod tests { fn esc_reset_intermediates() { const INPUT: &[u8] = b"\x1b[?2004l\x1b#8"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1612,7 +1489,7 @@ mod tests { fn params_buffer_filled_with_subparam() { const INPUT: &[u8] = b"\x1b[::::::::::::::::::::::::::::::::x\x1b"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1628,84 +1505,6 @@ mod tests { } } - #[cfg(not(feature = "std"))] - #[test] - fn build_with_fixed_size() { - const INPUT: &[u8] = b"\x1b[3;1\x1b[?1049h"; - let mut dispatcher = Dispatcher::default(); - let mut parser: Parser<30> = Parser::new_with_size(); - - parser.advance(&mut dispatcher, INPUT); - - assert_eq!(dispatcher.dispatched.len(), 1); - match &dispatcher.dispatched[0] { - Sequence::Csi(params, intermediates, ignore, _) => { - assert_eq!(intermediates, b"?"); - assert_eq!(params, &[[1049]]); - assert!(!ignore); - } - _ => panic!("expected csi sequence"), - } - } - - #[cfg(not(feature = "std"))] - #[test] - fn exceed_fixed_osc_buffer_size() { - const OSC_BUFFER_SIZE: usize = 32; - const NUM_BYTES: usize = OSC_BUFFER_SIZE + 100; - const INPUT_START: &[u8] = b"\x1b]52;"; - const INPUT_END: &[u8] = b"\x07"; - - let mut dispatcher = Dispatcher::default(); - let mut parser: Parser = Parser::new_with_size(); - - // Create valid OSC escape - parser.advance(&mut dispatcher, INPUT_START); - - // Exceed max buffer size - parser.advance(&mut dispatcher, &[b'a'; NUM_BYTES]); - - // Terminate escape for dispatch - parser.advance(&mut dispatcher, INPUT_END); - - assert_eq!(dispatcher.dispatched.len(), 1); - match &dispatcher.dispatched[0] { - Sequence::Osc(params, _) => { - assert_eq!(params.len(), 2); - assert_eq!(params[0], b"52"); - assert_eq!(params[1].len(), OSC_BUFFER_SIZE - params[0].len()); - for item in params[1].iter() { - assert_eq!(*item, b'a'); - } - } - _ => panic!("expected osc sequence"), - } - } - - #[cfg(not(feature = "std"))] - #[test] - fn fixed_size_osc_containing_string_terminator() { - const INPUT_START: &[u8] = b"\x1b]2;"; - const INPUT_MIDDLE: &[u8] = b"s\xe6\x9c\xab"; - const INPUT_END: &[u8] = b"\x1b\\"; - - let mut dispatcher = Dispatcher::default(); - let mut parser: Parser<5> = Parser::new_with_size(); - - parser.advance(&mut dispatcher, INPUT_START); - parser.advance(&mut dispatcher, INPUT_MIDDLE); - parser.advance(&mut dispatcher, INPUT_END); - - assert_eq!(dispatcher.dispatched.len(), 2); - match &dispatcher.dispatched[0] { - Sequence::Osc(params, false) => { - assert_eq!(params[0], b"2"); - assert_eq!(params[1], INPUT_MIDDLE); - } - _ => panic!("expected osc sequence"), - } - } - fn expect_opaque_sequence( input: &[u8], kind: OpaqueSequenceKind, @@ -1722,7 +1521,7 @@ mod tests { } let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, input); assert_eq!(dispatcher.dispatched, expected_dispatched); @@ -1792,7 +1591,7 @@ mod tests { fn parse_kitty_apc() { const INPUT: &[u8] = b"\x1b_Gf=24,s=10,v=20;Zm9v\x1b\\"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1831,7 +1630,7 @@ mod tests { // Only semicolons should separate control data from payload const INPUT: &[u8] = b"\x1b_Gf=32,s=10,v=20;AQIDBA==\x1b\\"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1882,7 +1681,7 @@ mod tests { b"\xF0\x9F\x8E\x89_\xF0\x9F\xA6\x80\xF0\x9F\xA6\x80_\xF0\x9F\x8E\x89"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1900,7 +1699,7 @@ mod tests { const INPUT: &[u8] = b"a\xEF\xBCb"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -1915,7 +1714,7 @@ mod tests { const INPUT: &[u8] = b"\xF0\x9F\x9A\x80"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, &INPUT[..1]); parser.advance(&mut dispatcher, &INPUT[1..2]); @@ -1936,7 +1735,7 @@ mod tests { const INPUT: &[u8] = b"\xC4\xB8\xF0\x9F\x8E\x89"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, &INPUT[..1]); parser.advance(&mut dispatcher, &INPUT[1..]); @@ -1951,7 +1750,7 @@ mod tests { const INPUT: &[u8] = b"a\xEF\xBCb"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, &INPUT[..1]); parser.advance(&mut dispatcher, &INPUT[1..2]); @@ -1969,7 +1768,7 @@ mod tests { const INPUT: &[u8] = b"\xE4\xBF\x99\xB5"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, &INPUT[..2]); parser.advance(&mut dispatcher, &INPUT[2..]); @@ -1983,7 +1782,7 @@ mod tests { const INPUT: &[u8] = b"\xD8\x1b012"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -2002,7 +1801,7 @@ mod tests { const INPUT: &[u8] = b"\x00\x1f\x80\x90\x98\x9b\x9c\x9d\x9e\x9fa"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); @@ -2025,7 +1824,7 @@ mod tests { const INPUT: &[u8] = b"\x18\x1a"; let mut dispatcher = Dispatcher::default(); - let mut parser = Parser::new(); + let mut parser = Parser::default(); parser.advance(&mut dispatcher, INPUT); diff --git a/copa/src/params.rs b/rio-backend/src/performer/parser/params.rs similarity index 98% rename from copa/src/params.rs rename to rio-backend/src/performer/parser/params.rs index 7a184479..56b6dc0b 100644 --- a/copa/src/params.rs +++ b/rio-backend/src/performer/parser/params.rs @@ -1,6 +1,6 @@ //! Fixed size parameters list with optional subparameters. -use core::fmt::{self, Debug, Formatter}; +use std::fmt::{self, Debug, Formatter}; pub(crate) const MAX_PARAMS: usize = 32; diff --git a/rio-proc-macros/Cargo.toml b/rio-proc-macros/Cargo.toml deleted file mode 100644 index 11f0fec2..00000000 --- a/rio-proc-macros/Cargo.toml +++ /dev/null @@ -1,15 +0,0 @@ -[package] -authors = ["Raphael Amorim "] -description = "Rio proc macros" -repository = { workspace = true } -name = "rio-proc-macros" -license = "MIT" -version = { workspace = true } -edition = { workspace = true } - -[lib] -proc-macro = true - -[dependencies] -proc-macro2 = "1.0.88" -quote = "1.0.37" diff --git a/rio-proc-macros/LICENSE b/rio-proc-macros/LICENSE deleted file mode 100644 index ae74a92e..00000000 --- a/rio-proc-macros/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2022-present Raphael Amorim - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. diff --git a/rio-proc-macros/src/lib.rs b/rio-proc-macros/src/lib.rs deleted file mode 100644 index 500a1f48..00000000 --- a/rio-proc-macros/src/lib.rs +++ /dev/null @@ -1,182 +0,0 @@ -// https://github.com/alacritty/vte/blob/master/vte_generate_state_changes/Cargo.toml -// By Christian Duerr - -#![deny(clippy::all, clippy::if_not_else, clippy::enum_glob_use)] - -extern crate proc_macro; - -use std::iter::Peekable; - -use proc_macro2::TokenTree::{Group, Literal, Punct}; -use proc_macro2::{token_stream, TokenStream, TokenTree}; -use quote::quote; - -/// Create a `const fn` which will return an array with all state changes. -#[proc_macro] -pub fn generate_state_changes(item: proc_macro::TokenStream) -> proc_macro::TokenStream { - // Convert from proc_macro -> proc_macro2 - let item: TokenStream = item.into(); - let mut iter = item.into_iter().peekable(); - - // Determine output function name - let fn_name = iter.next().unwrap(); - - // Separator between name and body with state changes - expect_punct(&mut iter, ','); - - // Create token stream to assign each state change to the array - let assignments_stream = states_stream(&mut iter); - - quote!( - const fn #fn_name() -> [[u8; 256]; 13] { - let mut state_changes = [[0; 256]; 13]; - - #assignments_stream - - state_changes - } - ) - .into() -} - -/// Generate the array assignment statements for all origin states. -fn states_stream(iter: &mut impl Iterator) -> TokenStream { - let mut states_stream = next_group(iter).into_iter().peekable(); - - // Loop over all origin state entries - let mut tokens = quote!(); - while states_stream.peek().is_some() { - // Add all mappings for this state - tokens.extend(state_entry_stream(&mut states_stream)); - - // Allow trailing comma - optional_punct(&mut states_stream, ','); - } - tokens -} - -/// Generate the array assignment statements for one origin state. -fn state_entry_stream(iter: &mut Peekable) -> TokenStream { - // Origin state name - let state = iter.next().unwrap(); - - // Token stream with all the byte->target mappings - let mut changes_stream = next_group(iter).into_iter().peekable(); - - let mut tokens = quote!(); - while changes_stream.peek().is_some() { - // Add next mapping for this state - tokens.extend(change_stream(&mut changes_stream, &state)); - - // Allow trailing comma - optional_punct(&mut changes_stream, ','); - } - tokens -} - -/// Generate the array assignment statement for a single byte->target mapping -/// for one state. -fn change_stream( - iter: &mut Peekable, - state: &TokenTree, -) -> TokenStream { - // Start of input byte range - let start = next_usize(iter); - - // End of input byte range - let end = if optional_punct(iter, '.') { - // Read inclusive end of range - expect_punct(iter, '.'); - expect_punct(iter, '='); - next_usize(iter) - } else { - // Without range, end is equal to start - start - }; - - // Separator between byte input range and output state - expect_punct(iter, '='); - expect_punct(iter, '>'); - - // Token stream with target state and action - let mut target_change_stream = next_group(iter).into_iter().peekable(); - - let mut tokens = quote!(); - while target_change_stream.peek().is_some() { - // Target state/action for all bytes in the range - let (target_state, target_action) = target_change(&mut target_change_stream); - - // Create a new entry for every byte in the range - for byte in start..=end { - tokens.extend(quote!( - state_changes[State::#state as usize][#byte] = - pack(State::#target_state, Action::#target_action); - )); - } - } - tokens -} - -/// Get next target state and action. -fn target_change(iter: &mut Peekable) -> (TokenTree, TokenTree) { - let target_state = iter.next().unwrap(); - - // Separator between state and action - expect_punct(iter, ','); - - let target_action = iter.next().unwrap(); - - (target_state, target_action) -} - -/// Check if next token matches specific punctuation. -fn optional_punct(iter: &mut Peekable, c: char) -> bool { - match iter.peek() { - Some(Punct(punct)) if punct.as_char() == c => iter.next().is_some(), - _ => false, - } -} - -/// Ensure next token matches specific punctuation. -/// -/// # Panics -/// -/// Panics if the punctuation does not match. -fn expect_punct(iter: &mut impl Iterator, c: char) { - match iter.next() { - Some(Punct(ref punct)) if punct.as_char() == c => (), - token => panic!("Expected punctuation '{c}', but got {token:?}"), - } -} - -/// Get next token as [`usize`]. -/// -/// # Panics -/// -/// Panics if the next token is not a [`usize`] in hex or decimal literal -/// format. -fn next_usize(iter: &mut impl Iterator) -> usize { - match iter.next() { - Some(Literal(literal)) => { - let literal = literal.to_string(); - if let Some(prefix) = literal.strip_prefix("0x") { - usize::from_str_radix(prefix, 16).unwrap() - } else { - literal.parse::().unwrap() - } - } - token => panic!("Expected literal, but got {token:?}"), - } -} - -/// Get next token as [`Group`]. -/// -/// # Panics -/// -/// Panics if the next token is not a [`Group`]. -fn next_group(iter: &mut impl Iterator) -> TokenStream { - match iter.next() { - Some(Group(group)) => group.stream(), - token => panic!("Expected group, but got {token:?}"), - } -}