From c97cbb4db9a3fa98d5dd301e9c55d62836dfd983 Mon Sep 17 00:00:00 2001 From: Simon Heimlicher Date: Fri, 24 Jul 2026 16:23:31 +0200 Subject: [PATCH] Skip escape stripping when a fragment holds no escape byte Component timing over 12 real transcript tails showed the escape-stripping regex costing 49.9 ms against 10.4 ms for case folding, while 238 of 240 fragments contained no ESC byte at all. A pattern anchored on ESC cannot match a string without one, so proving absence with a byte scan (11.5 ms) short-circuits the regex entirely on almost every fragment. Unlike the ASCII fast path this applies to all input, and it is the larger of the two wins: original 101.6 ms/round 1.00x + escape-absence guard 64.1 ms/round 1.58x + ASCII fast path 57.9 ms/round 1.75x The normalize tests now compare both shipped paths against a pristine copy of the original formulation rather than against each other, so neither optimization can drift from the semantics the matcher was built on. --- .../AgentDetection/AgentSessionResolver.swift | 12 +++++-- ...gentSessionFingerprintNormalizeTests.swift | 32 +++++++++++++++---- 2 files changed, 35 insertions(+), 9 deletions(-) diff --git a/supacode/Infrastructure/AgentDetection/AgentSessionResolver.swift b/supacode/Infrastructure/AgentDetection/AgentSessionResolver.swift index 68884831..a4caf94e 100644 --- a/supacode/Infrastructure/AgentDetection/AgentSessionResolver.swift +++ b/supacode/Infrastructure/AgentDetection/AgentSessionResolver.swift @@ -640,8 +640,16 @@ nonisolated enum AgentSessionFingerprintMatcher { /// Reference implementation. `normalize` must agree with it for every input; /// `AgentSessionFingerprintNormalizeTests` asserts that over a corpus. static func normalizeGeneral(_ value: String) -> String { - value - .replacing(#/\u{001B}\[[0-?]*[ -\/]*[@-~]/#, with: " ") + // A pattern anchored on ESC cannot match a string with no ESC byte, and + // nearly every transcript fragment has none. Proving absence with a byte + // scan is several times cheaper than letting the regex engine walk the + // whole string to reach the same conclusion. + let stripped = + value.utf8.contains(0x1B) + ? value.replacing(#/\u{001B}\[[0-?]*[ -\/]*[@-~]/#, with: " ") + : value + return + stripped .lowercased() .split(whereSeparator: \Character.isWhitespace) .joined(separator: " ") diff --git a/supacodeTests/AgentSessionFingerprintNormalizeTests.swift b/supacodeTests/AgentSessionFingerprintNormalizeTests.swift index c87ca26a..c8ee1462 100644 --- a/supacodeTests/AgentSessionFingerprintNormalizeTests.swift +++ b/supacodeTests/AgentSessionFingerprintNormalizeTests.swift @@ -47,12 +47,26 @@ struct AgentSessionFingerprintNormalizeTests { "mixed ascii and ünicode with spacing", ] - @Test func fastPathMatchesReferenceForEveryCorpusInput() { + /// The original formulation, before either the ASCII fast path or the + /// escape-absence guard. Both shipped paths must reproduce it exactly. + private static func pristine(_ value: String) -> String { + value + .replacing(#/\u{001B}\[[0-?]*[ -\/]*[@-~]/#, with: " ") + .lowercased() + .split(whereSeparator: \Character.isWhitespace) + .joined(separator: " ") + } + + @Test func bothPathsMatchTheOriginalForEveryCorpusInput() { for input in Self.corpus { + let expected = Self.pristine(input) + #expect( + AgentSessionFingerprintMatcher.normalize(input) == expected, + "normalize diverged for \(String(reflecting: input))" + ) #expect( - AgentSessionFingerprintMatcher.normalize(input) - == AgentSessionFingerprintMatcher.normalizeGeneral(input), - "normalize diverged from the reference for \(String(reflecting: input))" + AgentSessionFingerprintMatcher.normalizeGeneral(input) == expected, + "normalizeGeneral diverged for \(String(reflecting: input))" ) } } @@ -65,10 +79,14 @@ struct AgentSessionFingerprintNormalizeTests { for _ in 0..<2000 { let length = Int.random(in: 0...40, using: &generator) let input = String((0..