From 335d7da1621b534d2bd85f8d3b990cd6b8d4aece Mon Sep 17 00:00:00 2001 From: super Date: Fri, 17 Jul 2026 14:38:46 +0900 Subject: [PATCH] fix(import): strip non-breaking-space thousands separators in sanitize_number (#2538) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(import): strip non-breaking-space thousands separators in sanitize_number The French/Scandinavian number format ("1 234,56") only removed ASCII whitespace via \s, which does not match a non-breaking space (U+00A0) or a narrow no-break space (U+202F). Real-world European CSV exports use those Unicode spaces as the thousands separator, so such amounts failed the final numeric guard and were silently coerced to "", losing the value on import. The branch also skipped the non-numeric junk stripping that the other formats apply. Strip everything that isn't a digit, the decimal separator, or a minus sign in this branch, matching the else branch's behavior. Fixes #2537 * fix(import): strip only whitespace for space-delimited number format Addresses review feedback: the previous filter removed any non-numeric character, so a misconfigured/mixed row using US separators ("1,234.56") under the "1 234,56" format had its period dropped and its comma turned into a decimal point, silently importing 1.23456 (off by ~1000x) instead of being rejected. Strip only whitespace via \p{Space}, which matches ASCII, non-breaking (U+00A0), and narrow no-break (U+202F) spaces. Unexpected punctuation is left in place so the existing numeric guard still rejects malformed values. Add a regression test for the mixed-punctuation case and a docstring for sanitize_number. * fix(import): strip leading/trailing currency junk for space-delimited numbers Follow-up to review on #2538: the whitespace-only strip did not cover issue #2537's currency-suffix row ("1 234,56 kr") or a leading currency symbol ("€1 234,56"), which still failed the numeric guard and imported blank. Strip non-numeric junk only at the leading/trailing edges (currency symbols/codes), leaving interior characters untouched. This fully closes the #2537 repro table while preserving the earlier fix for the mixed US-format hazard: "1,234.56" keeps its interior period and is still rejected rather than parsed as 1.23456. Digits, the decimal separator, and a leading/trailing minus are preserved so signed values and the guard still behave correctly. Add regression tests for the currency prefix/suffix cases. * test(import): add U+202F narrow no-break-space regression case Complements the existing U+00A0 case so both Unicode thousands-separator variants named in the fix are covered. --- app/models/import.rb | 22 +++++- test/interfaces/import_interface_test.rb | 88 ++++++++++++++++++++++++ 2 files changed, 109 insertions(+), 1 deletion(-) diff --git a/app/models/import.rb b/app/models/import.rb index d2565a3e8..17e10dbaa 100644 --- a/app/models/import.rb +++ b/app/models/import.rb @@ -498,6 +498,13 @@ class Import < ApplicationRecord @parsed_csv = self.class.parse_csv_str(csv_content, col_sep: col_sep) end + # Normalizes a raw CSV numeric string into a plain, parseable decimal string + # based on the import's configured +number_format+ (thousands delimiter and + # decimal separator). Returns "" when the value is blank, the format is + # unknown, or the result is not a valid number. + # + # @param value [String, nil] the raw cell value from the CSV + # @return [String] a normalized number like "1234.56", or "" if invalid def sanitize_number(value) return "" if value.nil? @@ -509,7 +516,20 @@ class Import < ApplicationRecord # Handle French/Scandinavian format specially if format[:delimiter] == " " - sanitized = sanitized.gsub(/\s+/, "") # Remove all spaces first + # The thousands "space" can be an ASCII space, a non-breaking space + # (U+00A0) or a narrow no-break space (U+202F) depending on the locale + # or exporter. Ruby's \s does not match those Unicode spaces, so strip + # every kind of whitespace via the Unicode property. + sanitized = sanitized.gsub(/\p{Space}/, "") + + # Strip currency symbols/codes only at the leading/trailing edges (e.g. + # "€1 234,56" or "1 234,56 kr"). Interior characters are deliberately + # left in place so a misconfigured US-style value like "1,234.56" keeps + # its period and is rejected by the numeric guard below, rather than + # being silently reinterpreted as 1.23456. Digits, the separator, and a + # minus sign are preserved so signed values and the guard still work. + edge_junk = /\A[^\d#{Regexp.escape(format[:separator])}\-]+|[^\d#{Regexp.escape(format[:separator])}\-]+\z/ + sanitized = sanitized.gsub(edge_junk, "") else sanitized = sanitized.gsub(/[^\d#{Regexp.escape(format[:delimiter])}#{Regexp.escape(format[:separator])}\-]/, "") diff --git a/test/interfaces/import_interface_test.rb b/test/interfaces/import_interface_test.rb index 2b4759277..e30819594 100644 --- a/test/interfaces/import_interface_test.rb +++ b/test/interfaces/import_interface_test.rb @@ -112,6 +112,94 @@ module ImportInterfaceTest assert_equal "1234.56", row.amount end + test "parses French/Scandinavian format with non-breaking-space thousands separator" do + import = imports(:transaction) + import.update!( + number_format: "1 234,56", + amount_col_label: "amount", + date_col_label: "date", + name_col_label: "name", + date_format: "%m/%d/%Y" + ) + + # Real-world European CSV exports use a non-breaking space (U+00A0) as the + # thousands separator rather than a plain ASCII space. + csv_data = "date,amount,name\n01/01/2024,\"1 234,56\",Test" + import.update!(raw_file_str: csv_data) + import.generate_rows_from_csv + import.reload + + row = import.rows.first + assert_equal "1234.56", row.amount + end + + test "parses French/Scandinavian format with narrow no-break-space thousands separator" do + import = imports(:transaction) + import.update!( + number_format: "1 234,56", + amount_col_label: "amount", + date_col_label: "date", + name_col_label: "name", + date_format: "%m/%d/%Y" + ) + + # Some locales/exporters use a narrow no-break space (U+202F) as the + # thousands separator, which Ruby's \s also does not match. + csv_data = "date,amount,name\n01/01/2024,\"1 234,56\",Test" + import.update!(raw_file_str: csv_data) + import.generate_rows_from_csv + import.reload + + row = import.rows.first + assert_equal "1234.56", row.amount + end + + test "rejects US-punctuation values under the French/Scandinavian format" do + import = imports(:transaction) + import.update!( + number_format: "1 234,56", + amount_col_label: "amount", + date_col_label: "date", + name_col_label: "name", + date_format: "%m/%d/%Y" + ) + + # A misconfigured/mixed row using US separators ("1,234.56") must not be + # silently reinterpreted as 1.23456 under a space-delimited format; the + # unexpected period keeps it invalid so it surfaces as blank instead. + csv_data = "date,amount,name\n01/01/2024,\"1,234.56\",Test" + import.update!(raw_file_str: csv_data) + import.generate_rows_from_csv + import.reload + + row = import.rows.first + assert_equal "", row.amount + end + + test "strips leading/trailing currency junk under the French/Scandinavian format" do + import = imports(:transaction) + import.update!( + number_format: "1 234,56", + amount_col_label: "amount", + date_col_label: "date", + name_col_label: "name", + date_format: "%m/%d/%Y" + ) + + # Currency symbols/codes at the edges (issue #2537's "1 234,56 kr" row) are + # stripped; the amount still parses. Interior junk is not stripped (covered + # by the mixed-punctuation rejection test above). + csv_data = "date,amount,name\n" \ + "01/01/2024,\"1 234,56 kr\",Suffix\n" \ + "01/02/2024,\"€1 234,56\",Prefix" + import.update!(raw_file_str: csv_data) + import.generate_rows_from_csv + import.reload + + assert_equal "1234.56", import.rows.first.amount + assert_equal "1234.56", import.rows.second.amount + end + test "parses zero-decimal currency format correctly" do import = imports(:transaction) import.update!(