From b95e10e04a8d61fa04424a8456a110446cc061d1 Mon Sep 17 00:00:00 2001 From: David Hobley Date: Wed, 23 Sep 2026 08:38:59 +1000 Subject: [PATCH] fix(imap): normalise bare LF line endings in a fetched BODY[] literal MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Some servers hand back a message with the bare-LF line endings it was stored with. The MIME parser only recognises CRLF CRLF as the header/body separator, so such a body came back empty: the client showed the headers and nothing else. 2.2.1 added the same tolerance to TextMimeData, but a BODY[] literal arrives as bytes and goes through BinaryMimeData, which was not covered. The bytes cannot simply be decoded to a String to reuse that fix: the only charset that could be assumed at this point is wrong, because a message declares its charset per part in headers that have not been parsed yet. Decoding with Utf8Decoder(allowMalformed: true) turns every byte of a windows-1252 or latin-1 body into U+FFFD before the part's own charset= is ever read, so `Teší ma` would arrive as `Te ma` with no way to recover it. FetchParser therefore inserts the missing CR on the raw bytes, leaving every other byte exactly as the server sent it, and returns the input untouched (no copy) in the common conforming-CRLF case. --- lib/src/private/imap/fetch_parser.dart | 56 ++++++++++++++++- test/src/imap/fetch_parser_test.dart | 85 ++++++++++++++++++++++++++ 2 files changed, 138 insertions(+), 3 deletions(-) diff --git a/lib/src/private/imap/fetch_parser.dart b/lib/src/private/imap/fetch_parser.dart index 19ac29c3..0f8e2014 100644 --- a/lib/src/private/imap/fetch_parser.dart +++ b/lib/src/private/imap/fetch_parser.dart @@ -1,3 +1,5 @@ +import 'dart:typed_data'; + import '../../codecs/date_codec.dart'; import '../../codecs/mail_codec.dart'; import '../../imap/message_sequence.dart'; @@ -272,19 +274,67 @@ class FetchParser extends ResponseParser { } void _parseBodyFull(MimeMessage message, ImapValue bodyValue) { - //print("Parsing BODY[]\n[${bodyValue.value}]"); final data = bodyValue.data; final value = bodyValue.value; if (data != null) { - message.mimeData = BinaryMimeData(data, containsHeader: true); + // Normalise line endings to CRLF before parsing. Some IMAP servers + // preserve the original message's bare-LF line endings; the MIME parser + // only recognises \r\n\r\n as the header/body separator, so without + // normalisation the body is silently discarded. + // + // Done on the *bytes*. Decoding to a String first needs a charset, and + // the only one that can be assumed here is wrong: a message declares its + // charset per part, in headers this has not parsed yet. Running the + // literal through `Utf8Decoder(allowMalformed: true)` — which is what + // this did — turns every byte of a windows-1252 or latin-1 body into + // U+FFFD before the part's own `charset=` is ever read, so `Teší ma` + // arrived as `Te ma` and no later decode could recover it. + message.mimeData = BinaryMimeData( + _normaliseLineEndings(data), + containsHeader: true, + ); } else if (value != null) { message.mimeData = TextMimeData(value, containsHeader: true); - //print("Parsing BODY text \n$bodyText"); } // ensure all headers are set: message.parse(); } + static const int _cr = 13; + static const int _lf = 10; + + /// Gives every bare LF in [data] the CR the MIME parser expects, leaving + /// every other byte exactly as the server sent it. + /// + /// Returns [data] itself when there is nothing to do, which is the common + /// case — a conforming server sends CRLF already, and this runs over the + /// whole of every message body fetched. + static Uint8List _normaliseLineEndings(Uint8List data) { + var bareLineFeeds = 0; + for (var i = 0; i < data.length; i++) { + if (_isBareLineFeed(data, i)) { + bareLineFeeds++; + } + } + if (bareLineFeeds == 0) { + return data; + } + + final out = Uint8List(data.length + bareLineFeeds); + var index = 0; + for (var i = 0; i < data.length; i++) { + if (_isBareLineFeed(data, i)) { + out[index++] = _cr; + } + out[index++] = data[i]; + } + + return out; + } + + static bool _isBareLineFeed(Uint8List data, int i) => + data[i] == _lf && (i == 0 || data[i - 1] != _cr); + HeaderParseResult _parseBodyHeader( MimeMessage message, ImapValue headerValue, diff --git a/test/src/imap/fetch_parser_test.dart b/test/src/imap/fetch_parser_test.dart index bd96caa3..e0c53c18 100644 --- a/test/src/imap/fetch_parser_test.dart +++ b/test/src/imap/fetch_parser_test.dart @@ -7,6 +7,7 @@ import 'package:enough_mail/src/private/imap/all_parsers.dart'; import 'package:enough_mail/src/private/imap/imap_response.dart'; import 'package:enough_mail/src/private/imap/imap_response_line.dart'; import 'package:test/test.dart'; + // cSpell:disable void main() { @@ -1605,4 +1606,88 @@ Content-Transfer-Encoding: 8bit\r ); }); }); + + // Some servers hand back the message with the bare-LF line endings it was + // stored with. The MIME parser only recognises \r\n\r\n as the header/body + // separator, so those bodies came back empty — the reading pane showed the + // headers and nothing else. + // + // The first fix for that decoded the literal to a String to run replaceAll + // over it, which is what broke the 8bit tests above: it had to pick a + // charset before the part declaring one had been parsed. These two guard + // both halves at once, so neither can be fixed at the other's expense. + group('bare LF line endings', () { + MimeMessage? parseBodyFull(Uint8List messageData) { + final details = ImapResponse() + ..add( + ImapResponseLine('* 1 FETCH (UID 42 BODY[] {${messageData.length}}'), + ) + ..add(ImapResponseLine.raw(messageData)) + ..add(ImapResponseLine(')')); + final parser = FetchParser(isUidFetch: false); + final response = Response()..status = ResponseStatus.ok; + expect(parser.parseUntagged(details, response), true); + + return parser.parse(details, response)?.messages.first; + } + + test('a body separated by bare LFs is not lost', () { + const messageText = + 'Subject: Hello world\n' + 'Content-Type: text/plain; charset=us-ascii\n' + '\n' + 'the body\n'; + final message = parseBodyFull( + Uint8List.fromList(ascii.encode(messageText)), + ); + + expect(message?.decodeSubject(), 'Hello world'); + expect(message?.decodeContentText(), 'the body\r\n'); + }); + + test('a bare-LF message keeps its declared charset intact', () { + const codec = Windows1252Codec(); + const messageText = + 'Subject: Hello world\n' + 'Content-Type: text/plain; charset=windows-1252\n' + 'Content-Transfer-Encoding: 8bit\n' + '\n' + 'Teší ma, že vás spoznávam\n'; + final message = parseBodyFull( + Uint8List.fromList(codec.encode(messageText)), + ); + + expect(message?.decodeContentText(), 'Teší ma, že vás spoznávam\r\n'); + }); + + test('a conforming CRLF message is passed through untouched', () { + const messageText = + 'Subject: Hello world\r\n' + 'Content-Type: text/plain; charset=us-ascii\r\n' + '\r\n' + 'line one\r\n' + 'line two\r\n'; + final message = parseBodyFull( + Uint8List.fromList(ascii.encode(messageText)), + ); + + expect(message?.decodeContentText(), 'line one\r\nline two\r\n'); + }); + + test('a lone CR is not a line ending and gains no LF', () { + final messageText = + 'Subject: Hello world\r\n' + 'Content-Type: text/plain; charset=us-ascii\r\n' + '\r\n' + 'before${String.fromCharCode(13)}after\n'; + final message = parseBodyFull( + Uint8List.fromList(ascii.encode(messageText)), + ); + + expect( + message?.decodeContentText(), + 'before${String.fromCharCode(13)}after\r\n', + ); + }); + }); }