//! Fitting a value into a fixed-width column. //! //! Five functions with one subject: a listing prints a fixed-width table, the //! values going into it are records written by anybody, and something has to //! decide what to drop and how far to pad. All of these answers were copied //! instead of shared — [`ellipsize`] lived in `cmd::pr::read` and was reached //! as `crate::cmd::pr::read::ellipsize` from two other command families, //! [`day`] existed five times over as the same one-line body, and the padding //! was `{: String { if text.chars().count() > max { text.chars().take(max - 1).collect::() + "…" } else { text.to_string() } } /// Somebody else's text as one cell of a listing: cleaned, cut to `cut` /// characters, padded to `cells`. /// /// The three steps in the order they have to happen, at the four listings /// that were each spelling them out — `pr list`, `issue list`, `repo list` /// and `search`. Two of the steps were already copied per site; the third, /// [`crate::term::text::one_line`], is new, and putting it anywhere but /// first would be the bug it exists to prevent: [`ellipsize`] counts /// characters, so a title cut before the control characters come out has a /// budget partly spent on things that draw nothing, and [`width`] recognises /// escape sequences, so a hostile title padded before cleaning lines up /// perfectly with the rows either side of it. /// /// `cut` and `cells` are separate numbers because they are separate units — /// see the note at the top of this module — and the gap between them is the /// space between one column and the next. pub(crate) fn cell(text: &str, cut: usize, cells: usize) -> String { pad_to(&ellipsize(&crate::term::text::one_line(text), cut), cells) } /// How many terminal cells `text` will occupy when printed. /// /// The one question every fixed-width column asks, and the one `{: usize { let mut cells = 0; let mut chars = text.chars(); while let Some(c) = chars.next() { if c != '\x1b' { cells += c.width().unwrap_or(0); continue; } match chars.next() { // CSI: parameter and intermediate bytes, then one final byte. Some('[') => { for c in chars.by_ref() { if matches!(c, '\u{40}'..='\u{7e}') { break; } } } // OSC: a string terminated by BEL, or by ST spelled `ESC \`. Some(']') => { let mut after_esc = false; for c in chars.by_ref() { if c == '\x07' || (after_esc && c == '\\') { break; } after_esc = c == '\x1b'; } } // A two-character escape, or a trailing `ESC` with nothing after // it. Either way there is nothing more to skip. _ => {} } } cells } /// `text` followed by enough spaces to fill `cells` — `{: String { let mut padded = text.to_string(); padded.push_str(&" ".repeat(cells.saturating_sub(width(text)))); padded } /// [`pad_to`]'s mirror: enough spaces to fill `cells`, then `text` — `{:>N}` /// that counts what [`width`] counts. /// /// One caller, and it is the case that proves the point. `logs oauth` prints /// `{:>8}` over an already-painted invocation tag, and that was right only /// because `short_inv` happens to return exactly eight characters, so the /// padding it failed to apply was zero either way. pub(crate) fn pad_start(text: &str, cells: usize) -> String { let mut padded = " ".repeat(cells.saturating_sub(width(text))); padded.push_str(text); padded } /// The calendar date of an RFC 3339 timestamp, **in UTC**. /// /// Parsed rather than sliced. Taking the first ten characters printed the /// *writer's* local date, so a record written at `+03:00` and one written at /// `Z` could disagree by a day about the same instant — inside one listing, /// in a column whose only job is to let rows be compared. /// /// UTC is the choice because it is the only one that makes two rows /// comparable, which is what a column is for. The reader's local zone would /// make the same listing print differently on two machines and would make a /// piped listing depend on `TZ`; the writer's own offset is the bug being /// fixed. The visible consequence is real and intended: a record written at /// `+03:00` just before midnight now prints the previous day. See /// [Output contracts](crate::docs::output). /// /// A stamp that will not parse falls back to the first ten characters, which /// is what this function used to do unconditionally. That is not tidiness — /// listings have to keep rendering when one record is malformed, and an empty /// `createdAt` comes back empty rather than as a placeholder because old /// records really do carry one and the column is the caller's to fill. pub(crate) fn day(datetime: &str) -> String { match chrono::DateTime::parse_from_rfc3339(datetime) { Ok(stamp) => stamp.to_utc().date_naive().to_string(), Err(_) => datetime.chars().take(10).collect(), } } #[cfg(test)] mod tests { use super::*; #[test] fn ellipsize_only_cuts_what_is_too_long() { assert_eq!(ellipsize("short", 10), "short"); // Exactly at the limit is not too long. assert_eq!(ellipsize("0123456789", 10), "0123456789"); // One over: nine kept plus the marker, still ten cells. assert_eq!(ellipsize("0123456789a", 10), "012345678…"); assert_eq!(ellipsize("0123456789a", 10).chars().count(), 10); assert_eq!(ellipsize("", 10), ""); } /// Titles are user text and routinely contain multibyte characters, so /// the cut counts characters. Slicing bytes here would panic on a title /// whose 57th byte lands mid-character. #[test] fn ellipsize_counts_characters_not_bytes() { let emoji = "🧬".repeat(30); let cut = ellipsize(&emoji, 10); assert_eq!(cut.chars().count(), 10); assert_eq!(cut, "🧬".repeat(9) + "…"); // Combining marks count singly too — imperfect as display width goes, // but never a panic, which is the property being pinned. let accented = "é".repeat(12); assert_eq!(ellipsize(&accented, 5).chars().count(), 5); } /// The unit `ellipsize` cuts in is chars and the unit `width` measures in /// is cells, and they are allowed to disagree. Pinned so that a later /// reader does not "fix" one into the other without deciding to. #[test] fn the_cut_is_chars_and_the_measure_is_cells() { let cut = ellipsize(&"🧬".repeat(30), 10); assert_eq!(cut.chars().count(), 10); assert_eq!(width(&cut), 19, "nine wide glyphs plus a narrow ellipsis"); } #[test] fn ascii_is_one_cell_per_character() { assert_eq!(width(""), 0); assert_eq!(width("hello"), 5); } /// The defect this module was opened for: a CJK title counted by `char` /// is half the cells it prints in. #[test] fn cjk_is_two_cells_per_character() { let title = "日本語"; assert_eq!(title.chars().count(), 3); assert_eq!(width(title), 6); // And so a row padded on the char count would be three cells short. assert_eq!(pad_to(title, 10), "日本語 "); assert_eq!(width(&pad_to(title, 10)), 10); } #[test] fn an_emoji_is_two_cells() { assert_eq!(width("🧬"), 2); assert_eq!(width("a🧬b"), 4); assert_eq!(width(&pad_to("🧬", 6)), 6); } /// Padding applied after painting is the `logs pds` bug: the SGR bytes /// are counted, the string looks long, and no padding is applied at all. #[test] fn sgr_escapes_cost_no_cells() { let painted = crate::term::style::paint(true, crate::term::style::GOOD, "create"); assert!(painted.contains('\x1b'), "the test needs a real escape"); assert_eq!(width(&painted), 6); // Which is the whole point: a painted and an unpainted cell of the // same column land in the same place. assert_eq!(width(&pad_to(&painted, 8)), width(&pad_to("update", 8))); } /// A cell can carry an entire URL that occupies no columns at all, which /// is where measuring characters goes furthest wrong. #[test] fn an_osc_8_hyperlink_costs_only_its_text() { let linked = crate::term::hyperlink::wrap_when( true, "https://bsky.app/profile/permadeath.com", "@permadeath.com", ); assert!(linked.chars().count() > 50, "the URL really is in there"); assert_eq!(width(&linked), "@permadeath.com".chars().count()); assert_eq!(width(&pad_to(&linked, 20)), 20); // BEL terminates an OSC string too, and some emitters use it. assert_eq!(width("\x1b]8;;https://example.com\x07text\x1b]8;;\x07"), 4); } /// Rows line up. Stated as the property a column actually has to have, /// rather than only as the cell counts that produce it. #[test] fn a_padded_row_lines_up_whatever_is_in_the_first_column() { let painted = crate::term::style::paint(true, crate::term::style::BAD, "delete"); let linked = crate::term::hyperlink::wrap_when(true, "https://example.com/x", "🧬 x"); let rows = ["plain", "日本語", "🧬🧬", &painted, &linked]; let widths: Vec = rows .iter() .map(|cell| width(&format!("{} after", pad_to(cell, 12)))) .collect(); assert!( widths.windows(2).all(|w| w[0] == w[1]), "columns disagree: {widths:?}" ); } /// The rendering site, with the title out of the bug report: an OSC 8 /// hyperlink whose text and destination disagree, in a column that would /// otherwise have padded it to line up with the rows either side. #[test] fn a_listing_cell_carries_no_escape_out_of_a_record() { let hostile = "\x1b]8;;https://evil.example\x1b\\ok\x1b]8;;\x1b\\"; let row = cell(hostile, 54, 56); assert!(!row.contains('\x1b'), "{row:?}"); assert_eq!(row.trim_end(), "]8;;https://evil.example\\ok]8;;\\"); assert_eq!(width(&row), 56, "and it is still a 56-cell column"); // A screen-clearing title, and a title that would forge a second row // in a listing whose whole shape is one record per line. let cleared = cell("\x1b[2J\x1b[Hgone", 54, 56); assert!(!cleared.contains('\x1b'), "{cleared:?}"); let forged = cell("real\nforged", 54, 56); assert!(!forged.contains('\n'), "{forged:?}"); assert_eq!(forged.trim_end(), "real forged"); } /// Cleaning happens before the cut, so the character budget is spent on /// characters that draw. Cutting first would leave a title whose escapes /// ate the column and whose words did not fit. #[test] fn a_cell_is_cleaned_before_it_is_cut() { // Sixty escapes and four letters: everything a reader wants is past // the cut if the escapes are counted, and nothing is if they are not. let padded = cell(&("\x1b[0m".repeat(60) + "word"), 54, 56); assert_eq!(padded.trim_end(), "[0m".repeat(17) + "[0…"); assert_eq!(width(&padded), 56); } /// Too wide to fit comes back whole. A caller wanting a limit ellipsizes. #[test] fn padding_never_truncates() { assert_eq!(pad_to("far too long", 4), "far too long"); assert_eq!(pad_start("far too long", 4), "far too long"); } #[test] fn pad_start_fills_on_the_left() { assert_eq!(pad_start("7f3a", 8), " 7f3a"); let painted = crate::term::style::paint(true, crate::term::style::NOTE, "7f3a"); assert_eq!(width(&pad_start(&painted, 8)), 8); } /// The date out of a record's timestamp, in UTC, whatever offset the /// writer used. #[test] fn day_takes_the_date_off_an_iso_datetime() { assert_eq!(day("2026-08-05T04:31:07+03:00"), "2026-08-05"); assert_eq!(day("2026-08-05T01:34:10.355270Z"), "2026-08-05"); } /// The behaviour change, pinned: an offset that crosses midnight now /// prints the UTC day, not the writer's. Both stamps below are the same /// instant, and before this parsed they printed different dates. #[test] fn day_is_utc_not_the_writers_local_date() { assert_eq!(day("2026-08-05T01:31:07+03:00"), "2026-08-04"); assert_eq!(day("2026-08-04T22:31:07Z"), "2026-08-04"); assert_eq!( day("2026-08-05T01:31:07+03:00"), day("2026-08-04T22:31:07Z") ); // And the other way over the line. assert_eq!(day("2026-08-04T21:00:00-05:00"), "2026-08-05"); } /// A value that is not RFC 3339 comes back short instead of panicking or /// erroring, which is what keeps a listing rendering when one record is /// malformed. Records that list today really do carry these. #[test] fn day_degrades_on_anything_it_cannot_parse() { assert_eq!(day("2026-08-17"), "2026-08-17"); // Old records can carry an empty createdAt — one really does, in the // fixture — so this has to come back with something printable. assert_eq!(day(""), ""); assert_eq!(day("2026"), "2026"); assert_eq!(day("nope"), "nope"); // Counted in characters, not bytes — a multi-byte timestamp field is // nonsense, but slicing one by bytes is a panic in a listing. assert_eq!(day("héllo wörld!!"), "héllo wörl"); } }