Files
ai-job-search/tests/test_convert_salary_excel.py
T
OluwaJomilojuandClaude Sonnet 5 7f709eda57 fix(salary): require corroboration before accepting a header row (#415)
* fix(salary): require corroboration before accepting a header row

Header-row detection accepted the first row (of the first 10) where any
cell merely contained a company-pattern word - no check that the row
actually looked like a header. A source-citation row above the real
header table (standard in real Danish union/statistics exports, e.g.
"Kilde: ... opdelt efter arbejdsgiver ...") tripped it purely because
"arbejdsgiver" appeared in prose. The real header row then parsed as
data (its "Firma" cell became a bogus company), and every genuine
company silently lost all its salary data - exit 0, no warning.

A candidate row is now only accepted when a second cell also matches a
city/count/index pattern, and a sheet that ends up with zero detected
salary columns prints a warning instead of reporting success silently.

Fixes #414.

* fix(salary): require cross-cell corroboration, fall back for untyped columns

Two edge cases found in review of the corroboration fix:

- Same-cell corroboration wasn't enough: a citation sentence can pack a
  count-pattern word into the same sentence as the company-pattern one
  ("...opdelt efter arbejdsgiver, antal svar 1234"), which still passed
  the gate. Corroboration must now come from a different cell.

- The corroboration requirement itself broke sheets whose only real
  header has purely untyped salary columns (e.g. "Base pay 2025" /
  "Bonus 2025" - neither matches a known city/count/index pattern), so
  header detection found nothing at all. Falls back to the original
  any-cell-mentions-company rule when the strict pass finds no row in
  the first 10.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-03 08:19:08 +02:00

424 lines
17 KiB
Python

import io
import unittest
from contextlib import redirect_stderr
from types import SimpleNamespace
from tools.convert_salary_excel import (
INDEX_PATTERNS,
detect_column_type,
header_matches,
parse_numeric_cell,
parse_sheet,
)
class FakeWorksheet:
title = "Sheet1"
def __init__(self, rows):
self.rows = rows
def iter_rows(self, min_row=1, max_row=None, values_only=False):
rows = self.rows[min_row - 1:max_row]
for row in rows:
if values_only:
yield row
else:
yield [SimpleNamespace(value=value) for value in row]
def __getitem__(self, row_number):
return [SimpleNamespace(value=value) for value in self.rows[row_number - 1]]
class DetectColumnTypeTests(unittest.TestCase):
def test_index_headers_are_not_misclassified_as_count(self):
for header in ("Index", "Salary Index", "Engineering Index", "Median salary"):
with self.subTest(header=header):
self.assertEqual(detect_column_type(header), "index")
def test_single_letter_n_only_matches_as_a_token(self):
self.assertEqual(detect_column_type("Employee n"), "count")
self.assertEqual(detect_column_type("Engineering"), None)
def test_count_headers_still_match_common_labels(self):
for header in ("Count", "Engineering Count", "Antal medarbejdere"):
with self.subTest(header=header):
self.assertEqual(detect_column_type(header), "count")
def test_count_inside_word_does_not_make_count_header(self):
self.assertIsNone(detect_column_type("Accounting Total"))
self.assertEqual(detect_column_type("Accounting Index"), "index")
def test_danish_compound_headers_still_match(self):
self.assertEqual(detect_column_type("Lønindeks"), "index")
def test_compound_patterns_match_as_substring_but_others_do_not(self):
# A compound token (Danish "løn") matches inside a glued header word.
self.assertTrue(header_matches("lønindeks", INDEX_PATTERNS))
# A pattern that is not a compound token ("salary") only matches as a
# whole token, so it must not match inside an unrelated glued word.
self.assertFalse(header_matches("salaryindex", INDEX_PATTERNS))
self.assertTrue(header_matches("salary index", INDEX_PATTERNS))
def test_parse_sheet_preserves_category_name_with_letter_n(self):
ws = FakeWorksheet([
("Company", "Engineering Count", "Engineering Index"),
("Example Corp", 12, 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"]["engineering"], {"count": 12, "index": 105.5})
def test_parse_sheet_groups_accounting_count_index_pair(self):
ws = FakeWorksheet([
("Company", "Accounting Count", "Accounting Index"),
("Example Corp", 12, 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"]["accounting"], {"count": 12, "index": 105.5})
def test_parse_sheet_normalizes_paired_category_name_with_underscores(self):
ws = FakeWorksheet([
("Company", "Software Engineering Count", "Software Engineering Index"),
("Example Corp", 8, 110.0),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"]["software_engineering"], {"count": 8, "index": 110.0})
def test_parse_sheet_detects_company_column_with_token_header(self):
# Real-world salary sheets rarely use the bare token "Company";
# headers like "Company Name" / "Employer Name" must still be
# detected as the company column (previously silently skipped -> []).
for header in ("Company", "Company Name", "Employer Name"):
with self.subTest(header=header):
ws = FakeWorksheet([
(header, "Salary"),
("Example Corp", 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["company"], "Example Corp")
self.assertEqual(
companies[0]["categories"]["salary"], {"index": 105.5}
)
def test_parse_sheet_detects_city_column_with_token_header(self):
# City headers are matched with the same token-based header_matches()
# used for the company column, not exact string equality. Real-world
# sheets rarely use the bare token "City" or "Kommune" alone; headers
# like "City Name" / "City/Kommune" must still be detected as the city
# column (previously silently left as city_col=None -> empty city).
for header in ("City", "City Name", "Kommune", "City/Kommune"):
with self.subTest(header=header):
ws = FakeWorksheet([
("Company", header, "Salary"),
("Example Corp", "Aarhus", 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["city"], "Aarhus")
def test_parse_sheet_handles_ragged_rows(self):
# openpyxl's read_only mode yields ragged tuples for dimension-less
# workbooks: a row can be shorter than the header. A company row that
# omits its city and category cells must parse without an IndexError,
# be retained, and get an empty city.
ws = FakeWorksheet([
("Company", "City", "Engineering Count", "Engineering Index"),
("Example Corp",),
("Other Corp", "Aarhus", 12, 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 2)
self.assertEqual(companies[0]["company"], "Example Corp")
self.assertEqual(companies[0]["city"], "")
self.assertEqual(companies[0]["categories"], {})
self.assertEqual(companies[1]["categories"]["engineering"], {"count": 12, "index": 105.5})
def test_parse_sheet_skips_row_shorter_than_company_column(self):
# A ragged row that ends before the company column has no company cell
# at all; it must be skipped, not crash the parse.
ws = FakeWorksheet([
("Notes", "Company", "Salary Index"),
("stray",),
("", "Example Corp", 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["company"], "Example Corp")
def test_skips_free_text_column(self):
# A free-text "Notes" column must not become a bogus salary category.
ws = FakeWorksheet([
("Company", "Salary Index", "Notes"),
("Example Corp", 105.5, "good"),
])
companies = parse_sheet(ws)
self.assertIn("salary_index", companies[0]["categories"])
self.assertNotIn("notes", companies[0]["categories"])
def test_skips_numeric_identifier_column(self):
# A numeric "Id" column (employee id) must not be treated as a salary index.
ws = FakeWorksheet([
("Company", "Salary Index", "Id"),
("Example Corp", 105.5, 7),
])
companies = parse_sheet(ws)
self.assertIn("salary_index", companies[0]["categories"])
self.assertNotIn("id", companies[0]["categories"])
def test_keeps_numeric_salary_column(self):
# A genuine numeric salary column still produces a salary category.
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", 105.5),
])
companies = parse_sheet(ws)
self.assertIn("salary_index", companies[0]["categories"])
self.assertEqual(companies[0]["categories"]["salary_index"], {"index": 105.5})
def test_parse_sheet_accepts_comma_decimal_string_values(self):
# Locale-formatted Excel exports can carry numeric cells as strings.
# Danish decimal commas must not be silently dropped by float().
ws = FakeWorksheet([
("Company", "Engineering Count", "Engineering Index"),
("Example Corp", "12,0", "108,5"),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"]["engineering"],
{"count": 12, "index": 108.5},
)
def test_parse_sheet_accepts_danish_thousands_and_decimal_string(self):
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", "1.234,5"),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"]["salary_index"],
{"index": 1234.5},
)
def test_parse_sheet_skips_ambiguous_single_comma_thousands_string(self):
# In an English-locale export, "1,234" is probably 1234, but in a
# decimal-comma locale it could be 1.234. Preserve the old safe-skip
# behavior instead of guessing and writing a 1000x-wrong salary value.
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", "1,234"),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"], {})
def test_parse_sheet_skips_ambiguous_single_dot_thousands_string(self):
# "1.234" is the dot-side mirror of the comma guard above: in a
# decimal-dot locale it is 1.234, while a Danish export (whole
# thousands, no decimal comma, e.g. "60.000") means 1234/60000.
# float() used to write the 1000x-smaller value silently - the
# same never-guess policy must apply to both separators.
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", "1.234"),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"], {})
def test_parse_sheet_pairs_interleaved_count_index_columns_by_name(self):
ws = FakeWorksheet([
("Company", "Antal kvinder", "Antal mænd", "Kvinder indeks", "Mænd indeks"),
("Example Corp", 15, 20, 95.0, 108.0),
])
companies = parse_sheet(ws)
categories = companies[0]["categories"]
self.assertEqual(categories["kvinder"], {"count": 15, "index": 95.0})
self.assertEqual(categories["mænd"], {"count": 20, "index": 108.0})
def test_standalone_count_column_is_stored_as_count_not_index(self):
# A count column with no matching index column (e.g. a lone total
# headcount) is still count data. It must not be emitted as a salary
# index, which salary_lookup would render with a bogus "vs baseline"
# percentage. The paired category alongside it is unaffected.
ws = FakeWorksheet([
("Company", "Antal", "IT Count", "IT Index"),
("Example Corp", 250, 30, 108.5),
])
companies = parse_sheet(ws)
categories = companies[0]["categories"]
self.assertEqual(categories["antal"], {"count": 250})
self.assertEqual(categories["it"], {"count": 30, "index": 108.5})
def test_parse_sheet_non_adjacent_columns_no_cross_match(self):
ws = FakeWorksheet([
("Company", "Count_A", "Count_B", "Index_A", "Index_B"),
("Example Corp", 10, 20, 100.0, 200.0),
])
companies = parse_sheet(ws)
categories = companies[0]["categories"]
self.assertEqual(categories["a"], {"count": 10, "index": 100.0})
self.assertEqual(categories["b"], {"count": 20, "index": 200.0})
def test_parse_sheet_ignores_citation_row_mentioning_company_pattern_word(self):
# A title/source-citation row above the real header - standard in
# real Danish union/statistics exports - can contain a stray
# company-pattern word ("arbejdsgiver" = employer) in running prose.
# It must not be mistaken for the header: that misreads the real
# header row as data (producing a bogus "Firma" company) and drops
# every real company's salary data (issue #414).
ws = FakeWorksheet([
("Lønstatistik 2025",),
("Kilde: Medlemsundersøgelse opdelt efter arbejdsgiver og branche",),
(),
("Firma", "By", "Antal alle", "Lønindeks alle"),
("Novo Nordisk A/S", "Bagsværd", 500, 108.5),
("Ørsted A/S", "Fredericia", 200, 105.2),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 2)
self.assertEqual(companies[0]["company"], "Novo Nordisk A/S")
self.assertEqual(companies[0]["city"], "Bagsværd")
self.assertEqual(companies[0]["categories"]["alle"], {"count": 500, "index": 108.5})
self.assertEqual(companies[1]["company"], "Ørsted A/S")
def test_parse_sheet_rejects_citation_row_with_count_word_in_same_cell(self):
# Corroboration must come from a DIFFERENT cell than the company
# match. A single free-text sentence can pack both a company-pattern
# word and a count-pattern word together (e.g. "... opdelt efter
# arbejdsgiver, antal svar 1234") - same-cell corroboration must not
# be enough, or this citation row reintroduces the bogus-header bug.
ws = FakeWorksheet([
("Lønstatistik 2025",),
("Kilde: undersøgelse opdelt efter arbejdsgiver, antal svar 1234",),
(),
("Firma", "By", "Antal alle", "Lønindeks alle"),
("Novo Nordisk A/S", "Bagsværd", 500, 108.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["company"], "Novo Nordisk A/S")
self.assertEqual(companies[0]["categories"]["alle"], {"count": 500, "index": 108.5})
def test_parse_sheet_falls_back_when_no_row_has_cross_cell_corroboration(self):
# A header with only untyped salary columns (no header matches a
# known city/count/index pattern - "Base pay"/"Bonus" don't) has
# nothing to corroborate against in any row. The strict cross-cell
# check must fall back to the original any-cell-mentions-company
# rule rather than failing to find a header at all.
ws = FakeWorksheet([
("Company", "Base pay 2025", "Bonus 2025"),
("Example Corp", 55000, 5000),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["company"], "Example Corp")
self.assertEqual(companies[0]["categories"]["base_pay_2025"], {"index": 55000.0})
self.assertEqual(companies[0]["categories"]["bonus_2025"], {"index": 5000.0})
def test_parse_sheet_warns_when_no_salary_columns_detected(self):
# A header row with only company/city columns and no salary data
# is a strong signal something is wrong (a misdetected header row,
# or a sheet with no salary data at all) - it should be flagged,
# not silently reported as a successful conversion.
ws = FakeWorksheet([
("Company", "City"),
("Example Corp", "Aarhus"),
])
stderr = io.StringIO()
with redirect_stderr(stderr):
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"], {})
self.assertIn("No salary data columns detected", stderr.getvalue())
if __name__ == "__main__":
unittest.main()
class ParseNumericCellLocaleTests(unittest.TestCase):
# The separator that appears LAST is the decimal separator. Assuming
# European ("." thousands, "," decimal) for every both-separator string
# turned a US "1,234.56" into 1.23456 - a silent 1000x corruption that
# flowed into salary_data.json and negotiation advice.
def test_us_thousands_and_decimal_string(self):
self.assertEqual(parse_numeric_cell("1,234.56"), 1234.56)
def test_us_multiple_thousands_groups(self):
self.assertEqual(parse_numeric_cell("1,234,567.89"), 1234567.89)
def test_european_thousands_and_decimal_string(self):
self.assertEqual(parse_numeric_cell("1.234,56"), 1234.56)
def test_european_multiple_thousands_groups(self):
self.assertEqual(parse_numeric_cell("1.234.567,89"), 1234567.89)
class CompoundCategoryPairingTests(unittest.TestCase):
def test_parse_sheet_pairs_danish_compound_index_with_count(self):
# "Lønindeks alle" is *detected* as an index column via
# COMPOUND_PATTERNS, but the derived category name must also lose the
# compound word or it can never pair with "Antal alle" ("alle" vs
# "lønindeks alle") - exactly the locale the compound support exists for.
ws = FakeWorksheet([
("Firma", "Antal alle", "Lønindeks alle"),
("Example Corp", 12, 118.0),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"]["alle"],
{"count": 12, "index": 118.0},
)
def test_parse_sheet_sheet_level_us_locale_value(self):
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", "1,234.56"),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"]["salary_index"],
{"index": 1234.56},
)