[AI] Detect CSV file encoding and add a per-account encoding selector (#8924)
* [AI] Detect CSV file encoding and add a per-account encoding selector Read CSV/TSV files as raw bytes and decode them based on the detected encoding instead of always assuming UTF-8: the byte order mark, BOM-less UTF-16 (NUL byte analysis), strict UTF-8 validation, and a windows-1252 fallback that also decodes ISO-8859-1 content. A mostly-valid UTF-8 file with a few corrupted bytes still decodes as UTF-8 with replacement characters instead of falling back to windows-1252. Add an encoding selector to the CSV import options, persisted per account like the delimiter, for files in encodings that cannot be detected automatically (e.g. windows-1250, ISO-8859-2). Adds fixtures and tests for UTF-16 (with and without BOM), UTF-8 with BOM, corrupted UTF-8, NUL-padded UTF-8, and windows-1252 content. Fixes #6327 * Update VRT screenshots Auto-generated by VRT workflow PR: #8924 * Update VRT screenshots Auto-generated by VRT workflow PR: #8924 * [AI] Address review: require dominant NUL parity and translate encoding labels Require one parity to clearly dominate (>= 90% of NUL bytes) before detecting BOM-less UTF-16, so a UTF-8 file with dense contiguous NUL padding (even split across parities) is no longer misdetected as UTF-16. Adds a fixture with padding above the 10% density threshold. Wrap the encoding selector labels in t() per the repository's translated user-facing text requirement. * [AI] Drop iso-8859-1 from the CSV encoding selector TextDecoder resolves the iso-8859-1 label to the windows-1252 decoder per the WHATWG encoding spec, so the option was redundant and could mislead users into expecting true ISO-8859-1 decoding. windows-1252 already covers ISO-8859-1 content; keep the alias accepted in decodeCsvBytes with a clarifying comment and an alias test. * [AI] Make the iso-8859-1 alias test distinguish Windows-1252 The alias test now uses a fixture containing the 0x80 byte, which windows-1252 decodes as the euro sign while a true ISO-8859-1 decoder would produce a C1 control character, so the assertion actually verifies the windows-1252 alias semantics. * [AI] Simplify CSV encoding detection to BOM and explicit selection Follow the review suggestion: the byte order mark selects UTF-16 LE/BE and everything else decodes as UTF-8; other encodings are handled by the manual per-account selector this PR adds. Drop the speculative BOM-less UTF-16, windows-1252 fallback, and corrupted UTF-8 heuristics along with their fixtures and tests. --------- Co-authored-by: François Lafleur <[email protected]> Co-authored-by: github-actions[bot] <github-actions[bot]@users.noreply.github.com>
@@ -22,3 +22,4 @@ packages/mobile-client/android/gradlew.bat text eol=crlf
|
||||
# Denote all files that are truly binary and should not be modified.
|
||||
*.png binary
|
||||
*.jpg binary
|
||||
packages/loot-core/src/mocks/files/utf-16*.csv binary
|
||||
|
||||
|
Before Width: | Height: | Size: 166 KiB After Width: | Height: | Size: 167 KiB |
|
Before Width: | Height: | Size: 162 KiB After Width: | Height: | Size: 163 KiB |
|
Before Width: | Height: | Size: 163 KiB After Width: | Height: | Size: 164 KiB |
|
Before Width: | Height: | Size: 148 KiB After Width: | Height: | Size: 149 KiB |
|
Before Width: | Height: | Size: 148 KiB After Width: | Height: | Size: 149 KiB |
|
Before Width: | Height: | Size: 146 KiB After Width: | Height: | Size: 148 KiB |
@@ -176,6 +176,7 @@ type LastParse = {
|
||||
const parseOptionKeys = [
|
||||
'hasHeaderRow',
|
||||
'delimiter',
|
||||
'encoding',
|
||||
'fallbackMissingPayeeToMemo',
|
||||
'swapPayeeAndMemo',
|
||||
'skipStartLines',
|
||||
@@ -242,6 +243,9 @@ export function ImportTransactionsModal({
|
||||
prefs[`csv-delimiter-${accountId}`] ||
|
||||
(filename.endsWith('.tsv') ? '\t' : ','),
|
||||
);
|
||||
const [csvEncoding, setCsvEncoding] = useState(
|
||||
prefs[`csv-encoding-${accountId}`] || 'auto',
|
||||
);
|
||||
const [skipStartLines, setSkipStartLines] = useState(
|
||||
parseInt(prefs[`csv-skip-start-lines-${accountId}`], 10) || 0,
|
||||
);
|
||||
@@ -477,6 +481,7 @@ export function ImportTransactionsModal({
|
||||
const fileType = getFileType(originalFileName);
|
||||
const parseOptions = getParseOptions(fileType, {
|
||||
delimiter,
|
||||
encoding: csvEncoding,
|
||||
hasHeaderRow,
|
||||
skipStartLines,
|
||||
skipEndLines,
|
||||
@@ -509,6 +514,7 @@ export function ImportTransactionsModal({
|
||||
}, [
|
||||
originalFileName,
|
||||
delimiter,
|
||||
csvEncoding,
|
||||
hasHeaderRow,
|
||||
skipStartLines,
|
||||
skipEndLines,
|
||||
@@ -559,6 +565,7 @@ export function ImportTransactionsModal({
|
||||
const fileType = getFileType(res[0]);
|
||||
const parseOptions = getParseOptions(fileType, {
|
||||
delimiter,
|
||||
encoding: csvEncoding,
|
||||
hasHeaderRow,
|
||||
skipStartLines,
|
||||
skipEndLines,
|
||||
@@ -733,6 +740,7 @@ export function ImportTransactionsModal({
|
||||
[`csv-mappings-${accountId}`]: JSON.stringify(fieldMappings),
|
||||
});
|
||||
savePrefs({ [`csv-delimiter-${accountId}`]: delimiter });
|
||||
savePrefs({ [`csv-encoding-${accountId}`]: csvEncoding });
|
||||
savePrefs({ [`csv-has-header-${accountId}`]: String(hasHeaderRow) });
|
||||
savePrefs({
|
||||
[`csv-skip-start-lines-${accountId}`]: String(skipStartLines),
|
||||
@@ -1188,6 +1196,34 @@ export function ImportTransactionsModal({
|
||||
style={{ width: 50 }}
|
||||
/>
|
||||
</label>
|
||||
<label
|
||||
htmlFor="csv-encoding-select"
|
||||
style={{
|
||||
display: 'flex',
|
||||
flexDirection: 'row',
|
||||
gap: 5,
|
||||
alignItems: 'baseline',
|
||||
}}
|
||||
>
|
||||
<Trans>Encoding:</Trans>
|
||||
<Select
|
||||
id="csv-encoding-select"
|
||||
options={[
|
||||
['auto', t('Auto (detect)')],
|
||||
['utf-8', t('UTF-8')],
|
||||
['utf-16le', t('UTF-16 LE')],
|
||||
['utf-16be', t('UTF-16 BE')],
|
||||
['windows-1252', t('Windows-1252')],
|
||||
['windows-1250', t('Windows-1250')],
|
||||
['iso-8859-2', t('ISO-8859-2')],
|
||||
]}
|
||||
value={csvEncoding}
|
||||
onChange={value => {
|
||||
setCsvEncoding(value);
|
||||
}}
|
||||
style={{ width: 130 }}
|
||||
/>
|
||||
</label>
|
||||
<label
|
||||
htmlFor="csv-skip-start-lines"
|
||||
style={{
|
||||
@@ -1355,8 +1391,9 @@ export function ImportTransactionsModal({
|
||||
|
||||
function getParseOptions(fileType: string, options: ParseFileOptions = {}) {
|
||||
if (fileType === 'csv') {
|
||||
const { delimiter, hasHeaderRow, skipStartLines, skipEndLines } = options;
|
||||
return { delimiter, hasHeaderRow, skipStartLines, skipEndLines };
|
||||
const { delimiter, encoding, hasHeaderRow, skipStartLines, skipEndLines } =
|
||||
options;
|
||||
return { delimiter, encoding, hasHeaderRow, skipStartLines, skipEndLines };
|
||||
}
|
||||
if (isOfxFile(fileType)) {
|
||||
const { fallbackMissingPayeeToMemo, importNotes, swapPayeeAndMemo } =
|
||||
|
||||
|
Can't render this file because it contains an unexpected character in line 1 and column 4.
|
|
Can't render this file because it contains an unexpected character in line 1 and column 4.
|
@@ -0,0 +1,3 @@
|
||||
"Felhasználónév","Számlaszám","Könyvelés dátuma","Összeg","Devizanem","Partner név","Partner IBAN száma","Partner számlaszáma","Partner bankkódja","Könyvelési információk","Tranzakcióazonosító","Hitel azonosító","Közlemény","Partner címe","Küldő címe","Megjegyzés","Kedvenc","Dátum","Tranzakció dátuma és ideje","Kategória","Kártyaszám","Cím","Tranzakció elfogadásának típusa","Kártya típusa","Termék típusa","Digitalizált kártya száma","Mobilapplikáció","Tranzakció típusa","Számlaegyenleg"
|
||||
"tariff package name","11600006-XXXXXXXX-XXXXXXXX","2025.12.04","100","HUF","","","","","","","","","","","","0","","","","","","","","","","","",""
|
||||
"tariff package name","11600006-XXXXXXXX-XXXXXXXX","2025.12.01","100","HUF","","","","","","","","","","","","0","","","","","","","","","","","",""
|
||||
|
@@ -0,0 +1,2 @@
|
||||
"Date","Payee","Amount"
|
||||
"2025.12.04","Café €Rémy","100.25"
|
||||
|
@@ -0,0 +1,3 @@
|
||||
"Date","Payee","Amount"
|
||||
"2025.12.04","Café Rémy","100.25"
|
||||
"2025.12.05","Boulangerie Müller","-42.10"
|
||||
|
@@ -263,6 +263,127 @@ describe('File import', () => {
|
||||
expect(await getTransactions('one')).toMatchSnapshot();
|
||||
});
|
||||
|
||||
test('csv import works (utf-16le bank export)', async () => {
|
||||
const { errors, transactions } = await parseFile(
|
||||
__dirname + '/../../../mocks/files/utf-16le.csv',
|
||||
{ hasHeaderRow: true },
|
||||
);
|
||||
|
||||
expect(errors.length).toBe(0);
|
||||
expect(transactions).toHaveLength(2);
|
||||
expect(transactions[0]).toMatchObject({
|
||||
'Könyvelés dátuma': '2025.12.04',
|
||||
Összeg: '100',
|
||||
Devizanem: 'HUF',
|
||||
});
|
||||
});
|
||||
|
||||
test('csv import works (utf-16be)', async () => {
|
||||
const { errors, transactions } = await parseFile(
|
||||
__dirname + '/../../../mocks/files/utf-16be.csv',
|
||||
{ hasHeaderRow: true },
|
||||
);
|
||||
|
||||
expect(errors.length).toBe(0);
|
||||
expect(transactions).toHaveLength(2);
|
||||
expect(transactions[0]).toMatchObject({
|
||||
'Könyvelés dátuma': '2025.12.04',
|
||||
Összeg: '100',
|
||||
Devizanem: 'HUF',
|
||||
});
|
||||
});
|
||||
|
||||
test('csv import works (utf-8 with bom)', async () => {
|
||||
const { errors, transactions } = await parseFile(
|
||||
__dirname + '/../../../mocks/files/utf-8-bom.csv',
|
||||
{ hasHeaderRow: true },
|
||||
);
|
||||
|
||||
expect(errors.length).toBe(0);
|
||||
expect(transactions).toHaveLength(2);
|
||||
expect(transactions[0]).toMatchObject({
|
||||
'Könyvelés dátuma': '2025.12.04',
|
||||
Összeg: '100',
|
||||
Devizanem: 'HUF',
|
||||
});
|
||||
});
|
||||
|
||||
test('csv import skips start lines on utf-16le files', async () => {
|
||||
const { errors, transactions } = await parseFile(
|
||||
__dirname + '/../../../mocks/files/utf-16le.csv',
|
||||
{ hasHeaderRow: false, skipStartLines: 1 },
|
||||
);
|
||||
|
||||
expect(errors.length).toBe(0);
|
||||
expect(transactions).toHaveLength(2);
|
||||
const rows = transactions as string[][];
|
||||
expect(rows[0][0]).toBe('tariff package name');
|
||||
});
|
||||
|
||||
test('csv import respects manual encoding override', async () => {
|
||||
const { errors, transactions } = await parseFile(
|
||||
__dirname + '/../../../mocks/files/windows-1252.csv',
|
||||
{ hasHeaderRow: true, encoding: 'windows-1252' },
|
||||
);
|
||||
|
||||
expect(errors.length).toBe(0);
|
||||
expect(transactions).toHaveLength(2);
|
||||
expect(transactions[0]).toMatchObject({
|
||||
Date: '2025.12.04',
|
||||
Payee: 'Café Rémy',
|
||||
Amount: '100.25',
|
||||
});
|
||||
expect(transactions[1]).toMatchObject({
|
||||
Date: '2025.12.05',
|
||||
Payee: 'Boulangerie Müller',
|
||||
Amount: '-42.10',
|
||||
});
|
||||
});
|
||||
|
||||
test('csv import accepts iso-8859-1 as a windows-1252 alias', async () => {
|
||||
// TextDecoder resolves the iso-8859-1 label to windows-1252 per the
|
||||
// WHATWG encoding spec. The € byte (0x80) proves it: windows-1252 decodes
|
||||
// it as €, while a true ISO-8859-1 decoder would yield a C1 control char.
|
||||
const { errors, transactions } = await parseFile(
|
||||
__dirname + '/../../../mocks/files/windows-1252-euro.csv',
|
||||
{ hasHeaderRow: true, encoding: 'iso-8859-1' },
|
||||
);
|
||||
|
||||
expect(errors.length).toBe(0);
|
||||
expect(transactions).toHaveLength(1);
|
||||
expect(transactions[0]).toMatchObject({
|
||||
Date: '2025.12.04',
|
||||
Payee: 'Café €Rémy',
|
||||
Amount: '100.25',
|
||||
});
|
||||
});
|
||||
|
||||
test('csv import treats explicit auto encoding as detection', async () => {
|
||||
const { errors, transactions } = await parseFile(
|
||||
__dirname + '/../../../mocks/files/utf-16le.csv',
|
||||
{ hasHeaderRow: true, encoding: 'auto' },
|
||||
);
|
||||
|
||||
expect(errors.length).toBe(0);
|
||||
expect(transactions).toHaveLength(2);
|
||||
expect(transactions[0]).toMatchObject({
|
||||
'Könyvelés dátuma': '2025.12.04',
|
||||
Összeg: '100',
|
||||
Devizanem: 'HUF',
|
||||
});
|
||||
});
|
||||
|
||||
test('csv import rejects an invalid encoding', async () => {
|
||||
const { errors, transactions } = await parseFile(
|
||||
__dirname + '/../../../mocks/files/utf-8-bom.csv',
|
||||
{ hasHeaderRow: true, encoding: 'not-an-encoding' },
|
||||
);
|
||||
|
||||
expect(errors.length).toBe(1);
|
||||
expect(transactions).toHaveLength(0);
|
||||
expect(errors[0].message).toContain('Failed parsing');
|
||||
});
|
||||
|
||||
test('CAMT import respects ISO-8859-1 encoding', async () => {
|
||||
const { errors, transactions } = await parseFile(
|
||||
__dirname + '/../../../mocks/files/camt/camt.latin1.xml',
|
||||
|
||||
@@ -54,6 +54,31 @@ type StructuredTransaction = {
|
||||
category?: string | null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Decode raw CSV file bytes into a string. A user-provided encoding always
|
||||
* wins; otherwise the byte order mark selects UTF-16 LE/BE and everything
|
||||
* else decodes as UTF-8. Files in other encodings can be selected manually
|
||||
* in the import dialog.
|
||||
*/
|
||||
function decodeCsvBytes(bytes: Uint8Array, encoding = 'auto'): string {
|
||||
if (encoding !== 'auto') {
|
||||
// Per the WHATWG encoding spec, the iso-8859-1 label resolves to the
|
||||
// windows-1252 decoder; no browser provides a true ISO-8859-1 decoder,
|
||||
// and windows-1252 gives better results for legacy CSV content anyway.
|
||||
return new TextDecoder(encoding).decode(bytes);
|
||||
}
|
||||
|
||||
if (bytes[0] === 0xff && bytes[1] === 0xfe) {
|
||||
return new TextDecoder('utf-16le').decode(bytes);
|
||||
}
|
||||
|
||||
if (bytes[0] === 0xfe && bytes[1] === 0xff) {
|
||||
return new TextDecoder('utf-16be').decode(bytes);
|
||||
}
|
||||
|
||||
return new TextDecoder('utf-8').decode(bytes);
|
||||
}
|
||||
|
||||
// CSV files return raw data that are not guaranteed to be StructuredTransactions
|
||||
type CsvTransaction = Record<string, string> | string[];
|
||||
|
||||
@@ -73,6 +98,7 @@ export type ParseFileOptions = {
|
||||
skipStartLines?: number;
|
||||
skipEndLines?: number;
|
||||
importNotes?: boolean;
|
||||
encoding?: string;
|
||||
};
|
||||
|
||||
export async function parseFile(
|
||||
@@ -112,7 +138,18 @@ async function parseCSV(
|
||||
options: ParseFileOptions,
|
||||
): Promise<ParseFileResult> {
|
||||
const errors = Array<ParseError>();
|
||||
let contents = await fs.readFile(filepath);
|
||||
const bytes = await fs.readFile(filepath, 'binary');
|
||||
|
||||
let contents: string;
|
||||
try {
|
||||
contents = decodeCsvBytes(bytes, options.encoding);
|
||||
} catch (err) {
|
||||
errors.push({
|
||||
message: 'Failed parsing: ' + err.message,
|
||||
internal: err.message,
|
||||
});
|
||||
return { errors, transactions: [] };
|
||||
}
|
||||
|
||||
const skipStart = Math.max(0, options.skipStartLines || 0);
|
||||
const skipEnd = Math.max(0, options.skipEndLines || 0);
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
---
|
||||
category: Bugfix
|
||||
authors: [flafleur]
|
||||
---
|
||||
|
||||
Fix CSV transaction import failing on UTF-16 encoded bank export files, automatically detect their byte order mark, and add a per-account encoding selector for less common encodings
|
||||