// ByteLog.cpp — see ByteLog.h for design notes. #include "ByteLog.h" #include #include #include void ByteLog::scan(const QByteArray& bytes, ScannerState* state) { const auto* p = reinterpret_cast(bytes.constData()); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) const qsizetype n = bytes.size(); for (const qsizetype i : std::views::iota(qsizetype{0}, n)) { const unsigned char c = p[i]; // UTF-8 continuation tracking is independent of escape state: a multibyte // character can appear inside a sequence's payload bytes. if (state->utf8Remain > 0) { // Continuation bytes are 0x80-0xBF. A non-continuation byte here is // malformed UTF-8; treat it as ground (do not strand the counter). if ((c & 0xC0) == 0x80) { --state->utf8Remain; continue; } state->utf8Remain = 0; // fall through and process this byte normally. } if (c >= 0x80) { // Leading byte of a multibyte character. 2-byte (0xC0-0xDF): 1 // continuation; 3-byte (0xE0-0xEF): 2; 4-byte (0xF0-0xF7): 3. Longer // sequences (0xF8+) are not valid UTF-8; treat 0xF8-0xFB conservatively // as 4-continuation prefixes? No — keep strictly to UTF-8: invalid // lead bytes are single bytes. if (c >= 0xF0) { state->utf8Remain = (c <= 0xF7) ? 3 : 0; } else if (c >= 0xE0) { state->utf8Remain = 2; } else if (c >= 0xC0) { state->utf8Remain = 1; } // 0x80-0xBF here (stray continuation) is malformed; single byte. continue; } switch (state->esc) { case 0: // ground if (c == 0x1B) { // ESC state->esc = 1; } break; case 1: { // saw ESC — dispatch on the introducer switch (c) { case static_cast('['): state->esc = 2; // CSI break; case static_cast(']'): case static_cast('_'): case static_cast('P'): state->esc = 4; // OSC / APC / DCS: string, terminated by ST or BEL break; case 0x1B: break; // ESC ESC — still waiting for an introducer default: state->esc = 0; // two-byte escape, done break; } break; } case 2: // CSI: parameter/intermediate bytes, then final 0x40-0x7E if (c >= 0x40 && c <= 0x7E) { state->esc = 0; // final byte } break; case 3: // OSC terminated by BEL if (c == 0x07) { state->esc = 0; } break; case 4: // string sequence: waiting for ESC \ (ST) or BEL (OSC only) if (c == 0x07) { state->esc = 0; // BEL ends OSC } else if (c == 0x1B) { state->esc = 5; // maybe ST } break; case 5: // saw ESC inside a string sequence if (c == static_cast('\\')) { state->esc = 0; // ST — terminated } else if (c == 0x1B) { // stay waiting (ESC ESC …) } else { state->esc = 4; // still inside the string } break; default: state->esc = 0; break; } } } void ByteLog::noteBoundaryIfGround() { // Called after append(): the scanner state already covers every byte of // every record. The boundary "after the last record" is ground iff the // current state is ground. Boundaries only ever get added at the tail; // trimming removes from the front, keeping index alignment via headSeq. const bool ground = m_scanner.atGround(); m_groundAfter.push_back(ground); if (ground) { m_lastGroundIndex = static_cast(m_groundAfter.size()) - 1; } } quint64 ByteLog::append(quint8 tag, const QByteArray& payload) { const quint64 seq = m_tailSeq; ++m_tailSeq; m_payloadBytes += payload.size(); m_records.push_back(Record{.seq = seq, .tag = tag, .payload = payload}); scan(payload, &m_scanner); noteBoundaryIfGround(); trim(); return seq; } void ByteLog::trim() { if (m_payloadBytes <= kCapBytes) { return; } m_lastTrimForced = false; // Excess bytes that must go. const qint64 excess = m_payloadBytes - kCapBytes; const auto count = static_cast(m_records.size()); // Cumulative dropped bytes per front index. qsizetype dropThrough = -1; qsizetype forcedThrough = -1; qint64 dropped = 0; for (const qsizetype i : std::views::iota(qsizetype{0}, count)) { dropped += m_records[static_cast(i)].payload.size(); if (dropThrough < 0 && dropped >= excess) { forcedThrough = i; // minimal prefix that gets us under cap } // A ground boundary that also gets us under cap is the preferred stop. if (dropThrough < 0 && dropped >= excess && i <= m_lastGroundIndex && m_groundAfter[static_cast(i)]) { dropThrough = i; break; } } if (dropThrough < 0) { // No ground boundary gets us under cap. Drop the minimal prefix anyway // (unbounded memory is worse) and record the fact. m_lastTrimForced = true; ++m_forcedTrims; dropThrough = forcedThrough; } if (dropThrough < 0) { return; // cannot happen: cap exceeded implies at least one record } const qint64 actuallyDropped = [&] { qint64 d = 0; for (const qsizetype i : std::views::iota(qsizetype{0}, dropThrough + 1)) { d += m_records[static_cast(i)].payload.size(); } return d; }(); m_records.erase(m_records.begin(), m_records.begin() + dropThrough + 1); m_groundAfter.erase(m_groundAfter.begin(), m_groundAfter.begin() + dropThrough + 1); m_lastGroundIndex -= dropThrough + 1; m_headSeq += static_cast(dropThrough + 1); m_payloadBytes -= actuallyDropped; } QVector ByteLog::readFrom(quint64 fromSeq) const { QVector out; fromSeq = std::max(fromSeq, m_headSeq); // clamp to floor; caller detects via floor() for (const Record& r : m_records) { if (r.seq >= fromSeq) { out.push_back(r); } } return out; } QVector ByteLog::readFrom(quint64 fromSeq, quint64 maxRecords) const { QVector out; if (maxRecords == 0) { return out; } fromSeq = std::max(fromSeq, m_headSeq); // clamp to floor; caller detects via floor() out.reserve(static_cast(std::min( maxRecords, static_cast(m_records.size())))); for (const Record& r : m_records) { if (out.size() >= qsizetype(maxRecords)) { break; } if (r.seq >= fromSeq) { out.push_back(r); } } return out; }