fix(convert-salary-excel): reject ambiguous dot thousands separators (#326)

This commit is contained in:
Oscar Madera
2026-08-14 10:51:06 +02:00
committed by GitHub
parent 45d55a7452
commit 621ce5ab39
3 changed files with 32 additions and 0 deletions
+14
View File
@@ -11,6 +11,20 @@ prefer updating to a tagged release over pulling raw `master` (see
files a release touched; `python3 tools/check_upstream_updates.py` lists them with
per-file diff commands.
## [Unreleased]
### Fixed
- **`convert_salary_excel.py` no longer misreads whole-thousands cells from a Danish-locale
export** - a cell like `60.000` (thousands separator, no decimal comma) was handed to
`float()` and silently written as `60.0`, a 1000x-wrong salary in `salary_data.json` that
then rendered with a meaningless `vs baseline` percentage in `/apply`. The comma-side
mirror (`1,234`) was already guarded as ambiguous and skipped; the dot side had no guard,
and tests only pinned the both-separators form (`1.234,5`). `\d+\.\d{3}` is now rejected
the same way, so the shared never-guess policy applies to both separators and the rows in
between (e.g. `60.000,50`, `108,5`) keep parsing exactly as before. Pinned by
`tests/test_convert_salary_excel.py`.
## [1.5.0] - 2026-08-12
### Added
+15
View File
@@ -230,6 +230,21 @@ class DetectColumnTypeTests(unittest.TestCase):
self.assertEqual(companies[0]["categories"], {})
def test_parse_sheet_skips_ambiguous_single_dot_thousands_string(self):
# "1.234" is the dot-side mirror of the comma guard above: in a
# decimal-dot locale it is 1.234, while a Danish export (whole
# thousands, no decimal comma, e.g. "60.000") means 1234/60000.
# float() used to write the 1000x-smaller value silently - the
# same never-guess policy must apply to both separators.
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", "1.234"),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"], {})
def test_parse_sheet_pairs_interleaved_count_index_columns_by_name(self):
ws = FakeWorksheet([
("Company", "Antal kvinder", "Antal mænd", "Kvinder indeks", "Mænd indeks"),
+3
View File
@@ -70,6 +70,9 @@ def parse_numeric_cell(value):
if re.fullmatch(r"[+-]?\d+,\d{3}", text):
raise ValueError("ambiguous comma separator")
text = text.replace(",", ".")
elif "." in text:
if re.fullmatch(r"[+-]?\d+\.\d{3}", text):
raise ValueError("ambiguous dot separator")
return float(text)