Files
workavia-mail-front/core/lib/utils/string_convert.dart
T
Théo Poizat a9cfe86a59 Use html parser instead of regex to clean composer input
Before sending input to the LLM, we clean the composer input to
remove html tags. We relied on regex that we not complete enough and
could be bypassed easily. So let's use the html parser available in
Dart instead.

I also added some tests.
2025-12-19 19:03:39 +07:00

228 lines
6.8 KiB
Dart

import 'dart:convert';
import 'dart:typed_data';
import 'package:core/utils/app_logger.dart';
import 'package:core/domain/exceptions/string_exception.dart';
import 'package:core/utils/mail/named_address.dart';
import 'package:html/dom.dart';
import 'package:html/parser.dart';
import 'package:http_parser/http_parser.dart';
class StringConvert {
static const String emailSeparatorPattern = r'[ ,;]+';
static String? writeEmptyToNull(String text) {
if (text.isEmpty) return null;
return text;
}
static String writeNullToEmpty(String? text) {
return text ?? '';
}
static String decodeBase64ToString(String text) {
try {
return utf8.decode(base64Decode(text));
} catch (e) {
logError('StringConvert::decodeBase64ToString:Exception = $e');
return text;
}
}
static List<String> extractStrings(String input, String separatorPattern) {
try {
// Check if the input is URL encoded
if (input.contains('%')) {
input = Uri.decodeComponent(input); // Decode URL encoding
}
// Efficient Base64 validation: Check length and minimal regex match
if (input.length % 4 == 0 && input.contains(RegExp(r'^[A-Za-z0-9+/=]+$'))) {
try {
input = utf8.decode(base64.decode(input)); // Decode Base64 encoding
} catch (_) {
// Ignore if decoding fails
}
}
final RegExp separator = RegExp(separatorPattern);
final listStrings = input
.replaceAll('\n', ' ')
.split(separator)
.map((value) => value.trim())
.where((value) => value.isNotEmpty)
.toList();
log('StringConvert::extractStrings:listStrings = $listStrings');
return listStrings;
} catch (e) {
return [];
}
}
static List<String> extractEmailAddress(String input) =>
extractStrings(input, emailSeparatorPattern);
static String decodeFromBytes(
Uint8List bytes, {
required String? charset,
bool isHtml = false,
}) {
if (isHtml) {
return utf8.decode(bytes);
} else if (charset == null) {
throw const NullCharsetException();
} else if (charset.toLowerCase().contains('utf-8')) {
return utf8.decode(bytes);
} else if (charset.toLowerCase().contains('latin')) {
return latin1.decode(bytes);
} else if (charset.toLowerCase().contains('ascii')) {
return ascii.decode(bytes);
} else {
throw const UnsupportedCharsetException();
}
}
static String toUrlScheme(String hostScheme) {
return '$hostScheme://';
}
static Uint8List convertBase64ImageTagToBytes(String base64ImageTag) {
if (!base64ImageTag.contains('base64,')) {
throw ArgumentError('The string is not valid Base64 data from an <img> tag.');
}
final base64Data = base64ImageTag.split(',').last;
return base64Decode(base64Data);
}
static MediaType? getMediaTypeFromBase64ImageTag(String base64ImageTag) {
try {
if (!base64ImageTag.startsWith("data:") || !base64ImageTag.contains(";base64,")) {
return null;
}
final mimeType = base64ImageTag.split(";")[0].split(":")[1];
log('StringConvert::getMediaTypeFromBase64ImageTag:mimeType = $mimeType');
return MediaType.parse(mimeType);
} catch (e) {
logError('StringConvert::getMimeTypeFromBase64ImageTag:Exception = $e');
return null;
}
}
static String getContentOriginal(String content) {
try {
final emailDocument = parse(content);
final contentOriginal = emailDocument.body?.innerHtml ?? content;
return contentOriginal;
} catch (e) {
logError('StringConvert::getContentOriginal:Exception = $e');
return content;
}
}
/// Checks if the given text is a table (supports Markdown or ASCII art format).
static bool isTextTable(String text) {
final lines =
text.split('\n').where((line) => line.trim().isNotEmpty).toList();
if (lines.length < 2) return false;
// Check for Markdown table
final mdSeparatorRegex = RegExp(r'^\|?(\s*:?-+:?\s*\|)+(\s*:?-+:?\s*)\|?$');
final isMarkdown = lines.any((line) => mdSeparatorRegex.hasMatch(line));
// Check for ASCII art table (with borders made of +, -, |, /, \, =)
final asciiArtRegex = RegExp(r'[\+\-\|/\\=]');
final isAsciiArt = lines.every((line) =>
line.contains(asciiArtRegex) && line.split(asciiArtRegex).length > 1);
log('StringConvert::isTextTable:isMarkdown = $isMarkdown | isAscii = $isAsciiArt');
return isMarkdown || isAsciiArt;
}
static List<NamedAddress> extractNamedAddresses(String input) {
try {
if (input.contains('%')) {
input = Uri.decodeComponent(input);
}
if (input.length % 4 == 0 &&
input.contains(RegExp(r'^[A-Za-z0-9+/=]+$'))) {
try {
input = utf8.decode(base64.decode(input));
} catch (_) {}
}
input = input.replaceAll('\n', ' ');
final results = <NamedAddress>[];
final pattern = RegExp(r'''(?:"([^"]+)"|'([^']+)'|)\s*<([^>]+)>''');
int currentIndex = 0;
final matches = pattern.allMatches(input).toList();
for (final match in matches) {
if (match.start > currentIndex) {
final between = input.substring(currentIndex, match.start);
results.addAll(_splitPlainAddresses(between, emailSeparatorPattern));
}
final name = match.group(1) ?? match.group(2) ?? '';
final email = match.group(3) ?? '';
results.add(NamedAddress(name: name.trim(), address: email.trim()));
currentIndex = match.end;
}
if (currentIndex < input.length) {
final tail = input.substring(currentIndex);
results.addAll(_splitPlainAddresses(tail, emailSeparatorPattern));
}
log('StringConvert::extractNamedAddresses:results = $results');
return results;
} catch (_) {
return [];
}
}
static List<NamedAddress> _splitPlainAddresses(
String input,
String emailSeparatorPattern,
) {
final separator = RegExp(emailSeparatorPattern);
return input
.split(separator)
.map((e) => e.trim())
.where((e) => e.isNotEmpty)
.map((e) => NamedAddress(name: '', address: e))
.toList();
}
static String convertHtmlContentToTextContent(String htmlContent) {
final document = parse(htmlContent);
// Each paragraph is surrounded by block tags so we add a /n for each block tag
// Even <br> are surrounded by block tags so we can ignore <br> and treat them
// as paragraph
const blockTags = [
'p',
'div',
'li',
'section',
'article',
'header',
'footer',
'h1','h2','h3','h4','h5','h6'
];
for (final tag in blockTags) {
document.querySelectorAll(tag).forEach((element) {
element.append(Text('\n'));
});
}
final String textContent = document.body?.text ?? '';
return textContent.trim();
}
}