mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 00:26:26 +00:00
fix(convert-salary-excel): reject ambiguous dot thousands separators (#326)
This commit is contained in:
@@ -11,6 +11,20 @@ prefer updating to a tagged release over pulling raw `master` (see
|
|||||||
files a release touched; `python3 tools/check_upstream_updates.py` lists them with
|
files a release touched; `python3 tools/check_upstream_updates.py` lists them with
|
||||||
per-file diff commands.
|
per-file diff commands.
|
||||||
|
|
||||||
|
## [Unreleased]
|
||||||
|
|
||||||
|
### Fixed
|
||||||
|
|
||||||
|
- **`convert_salary_excel.py` no longer misreads whole-thousands cells from a Danish-locale
|
||||||
|
export** - a cell like `60.000` (thousands separator, no decimal comma) was handed to
|
||||||
|
`float()` and silently written as `60.0`, a 1000x-wrong salary in `salary_data.json` that
|
||||||
|
then rendered with a meaningless `vs baseline` percentage in `/apply`. The comma-side
|
||||||
|
mirror (`1,234`) was already guarded as ambiguous and skipped; the dot side had no guard,
|
||||||
|
and tests only pinned the both-separators form (`1.234,5`). `\d+\.\d{3}` is now rejected
|
||||||
|
the same way, so the shared never-guess policy applies to both separators and the rows in
|
||||||
|
between (e.g. `60.000,50`, `108,5`) keep parsing exactly as before. Pinned by
|
||||||
|
`tests/test_convert_salary_excel.py`.
|
||||||
|
|
||||||
## [1.5.0] - 2026-08-12
|
## [1.5.0] - 2026-08-12
|
||||||
|
|
||||||
### Added
|
### Added
|
||||||
|
|||||||
@@ -230,6 +230,21 @@ class DetectColumnTypeTests(unittest.TestCase):
|
|||||||
|
|
||||||
self.assertEqual(companies[0]["categories"], {})
|
self.assertEqual(companies[0]["categories"], {})
|
||||||
|
|
||||||
|
def test_parse_sheet_skips_ambiguous_single_dot_thousands_string(self):
|
||||||
|
# "1.234" is the dot-side mirror of the comma guard above: in a
|
||||||
|
# decimal-dot locale it is 1.234, while a Danish export (whole
|
||||||
|
# thousands, no decimal comma, e.g. "60.000") means 1234/60000.
|
||||||
|
# float() used to write the 1000x-smaller value silently - the
|
||||||
|
# same never-guess policy must apply to both separators.
|
||||||
|
ws = FakeWorksheet([
|
||||||
|
("Company", "Salary Index"),
|
||||||
|
("Example Corp", "1.234"),
|
||||||
|
])
|
||||||
|
|
||||||
|
companies = parse_sheet(ws)
|
||||||
|
|
||||||
|
self.assertEqual(companies[0]["categories"], {})
|
||||||
|
|
||||||
def test_parse_sheet_pairs_interleaved_count_index_columns_by_name(self):
|
def test_parse_sheet_pairs_interleaved_count_index_columns_by_name(self):
|
||||||
ws = FakeWorksheet([
|
ws = FakeWorksheet([
|
||||||
("Company", "Antal kvinder", "Antal mænd", "Kvinder indeks", "Mænd indeks"),
|
("Company", "Antal kvinder", "Antal mænd", "Kvinder indeks", "Mænd indeks"),
|
||||||
|
|||||||
@@ -70,6 +70,9 @@ def parse_numeric_cell(value):
|
|||||||
if re.fullmatch(r"[+-]?\d+,\d{3}", text):
|
if re.fullmatch(r"[+-]?\d+,\d{3}", text):
|
||||||
raise ValueError("ambiguous comma separator")
|
raise ValueError("ambiguous comma separator")
|
||||||
text = text.replace(",", ".")
|
text = text.replace(",", ".")
|
||||||
|
elif "." in text:
|
||||||
|
if re.fullmatch(r"[+-]?\d+\.\d{3}", text):
|
||||||
|
raise ValueError("ambiguous dot separator")
|
||||||
return float(text)
|
return float(text)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user