diff --git a/app/models/import.rb b/app/models/import.rb index d2565a3e8..17e10dbaa 100644 --- a/app/models/import.rb +++ b/app/models/import.rb @@ -498,6 +498,13 @@ class Import < ApplicationRecord @parsed_csv = self.class.parse_csv_str(csv_content, col_sep: col_sep) end + # Normalizes a raw CSV numeric string into a plain, parseable decimal string + # based on the import's configured +number_format+ (thousands delimiter and + # decimal separator). Returns "" when the value is blank, the format is + # unknown, or the result is not a valid number. + # + # @param value [String, nil] the raw cell value from the CSV + # @return [String] a normalized number like "1234.56", or "" if invalid def sanitize_number(value) return "" if value.nil? @@ -509,7 +516,20 @@ class Import < ApplicationRecord # Handle French/Scandinavian format specially if format[:delimiter] == " " - sanitized = sanitized.gsub(/\s+/, "") # Remove all spaces first + # The thousands "space" can be an ASCII space, a non-breaking space + # (U+00A0) or a narrow no-break space (U+202F) depending on the locale + # or exporter. Ruby's \s does not match those Unicode spaces, so strip + # every kind of whitespace via the Unicode property. + sanitized = sanitized.gsub(/\p{Space}/, "") + + # Strip currency symbols/codes only at the leading/trailing edges (e.g. + # "€1 234,56" or "1 234,56 kr"). Interior characters are deliberately + # left in place so a misconfigured US-style value like "1,234.56" keeps + # its period and is rejected by the numeric guard below, rather than + # being silently reinterpreted as 1.23456. Digits, the separator, and a + # minus sign are preserved so signed values and the guard still work. + edge_junk = /\A[^\d#{Regexp.escape(format[:separator])}\-]+|[^\d#{Regexp.escape(format[:separator])}\-]+\z/ + sanitized = sanitized.gsub(edge_junk, "") else sanitized = sanitized.gsub(/[^\d#{Regexp.escape(format[:delimiter])}#{Regexp.escape(format[:separator])}\-]/, "") diff --git a/test/interfaces/import_interface_test.rb b/test/interfaces/import_interface_test.rb index 2b4759277..e30819594 100644 --- a/test/interfaces/import_interface_test.rb +++ b/test/interfaces/import_interface_test.rb @@ -112,6 +112,94 @@ module ImportInterfaceTest assert_equal "1234.56", row.amount end + test "parses French/Scandinavian format with non-breaking-space thousands separator" do + import = imports(:transaction) + import.update!( + number_format: "1 234,56", + amount_col_label: "amount", + date_col_label: "date", + name_col_label: "name", + date_format: "%m/%d/%Y" + ) + + # Real-world European CSV exports use a non-breaking space (U+00A0) as the + # thousands separator rather than a plain ASCII space. + csv_data = "date,amount,name\n01/01/2024,\"1 234,56\",Test" + import.update!(raw_file_str: csv_data) + import.generate_rows_from_csv + import.reload + + row = import.rows.first + assert_equal "1234.56", row.amount + end + + test "parses French/Scandinavian format with narrow no-break-space thousands separator" do + import = imports(:transaction) + import.update!( + number_format: "1 234,56", + amount_col_label: "amount", + date_col_label: "date", + name_col_label: "name", + date_format: "%m/%d/%Y" + ) + + # Some locales/exporters use a narrow no-break space (U+202F) as the + # thousands separator, which Ruby's \s also does not match. + csv_data = "date,amount,name\n01/01/2024,\"1 234,56\",Test" + import.update!(raw_file_str: csv_data) + import.generate_rows_from_csv + import.reload + + row = import.rows.first + assert_equal "1234.56", row.amount + end + + test "rejects US-punctuation values under the French/Scandinavian format" do + import = imports(:transaction) + import.update!( + number_format: "1 234,56", + amount_col_label: "amount", + date_col_label: "date", + name_col_label: "name", + date_format: "%m/%d/%Y" + ) + + # A misconfigured/mixed row using US separators ("1,234.56") must not be + # silently reinterpreted as 1.23456 under a space-delimited format; the + # unexpected period keeps it invalid so it surfaces as blank instead. + csv_data = "date,amount,name\n01/01/2024,\"1,234.56\",Test" + import.update!(raw_file_str: csv_data) + import.generate_rows_from_csv + import.reload + + row = import.rows.first + assert_equal "", row.amount + end + + test "strips leading/trailing currency junk under the French/Scandinavian format" do + import = imports(:transaction) + import.update!( + number_format: "1 234,56", + amount_col_label: "amount", + date_col_label: "date", + name_col_label: "name", + date_format: "%m/%d/%Y" + ) + + # Currency symbols/codes at the edges (issue #2537's "1 234,56 kr" row) are + # stripped; the amount still parses. Interior junk is not stripped (covered + # by the mixed-punctuation rejection test above). + csv_data = "date,amount,name\n" \ + "01/01/2024,\"1 234,56 kr\",Suffix\n" \ + "01/02/2024,\"€1 234,56\",Prefix" + import.update!(raw_file_str: csv_data) + import.generate_rows_from_csv + import.reload + + assert_equal "1234.56", import.rows.first.amount + assert_equal "1234.56", import.rows.second.amount + end + test "parses zero-decimal currency format correctly" do import = imports(:transaction) import.update!(