Files
ai-job-search/tests/test_convert_salary_excel.py
T
Ayobami Adegoke 968fb1bd7e fix(salary): pair a bare Count/Index column pair instead of splitting it into two categories (#470)
The pairing loop required a non-empty derived category name on both
sides, but a header with no category word - "Count" + "Index", Danish
"Antal" + "Lønindeks" - strips to an empty name, so the simplest layout
the README advertises ("auto-pairs count/index columns") came out as two
unrelated standalone categories:

  {"count": {"count": 500}, "index": {"index": 108.5}}

salary_lookup then rendered a "Count  500  N/A*" row above an
"Index  -  108.5" row, and its footnote read the N/A* as "too few
employees to publish (privacy)" - a false statement about a company whose
headcount is in the file, shown during /apply's salary step. Any suffix
("Antal alle") made pairing work, which is why the shipped tests, all
suffixed, never saw it.

Pair on equal derived names, empty included, and give the nameless pair
the README's top-level category name (all_employees). A bare "Antal" with
no bare index column still stays a standalone count; named pairs
alongside are untouched.

Four new cases in test_convert_salary_excel.py, one of them rendering the
converter's output through salary_lookup.format_entry; all four fail
against the old pairing rule.
2026-09-16 06:47:45 +02:00

503 lines
20 KiB
Python

import io
import unittest
from contextlib import redirect_stderr
from types import SimpleNamespace
from salary_lookup import format_entry
from tools.convert_salary_excel import (
INDEX_PATTERNS,
detect_column_type,
header_matches,
parse_numeric_cell,
parse_sheet,
)
class FakeWorksheet:
title = "Sheet1"
def __init__(self, rows):
self.rows = rows
def iter_rows(self, min_row=1, max_row=None, values_only=False):
rows = self.rows[min_row - 1:max_row]
for row in rows:
if values_only:
yield row
else:
yield [SimpleNamespace(value=value) for value in row]
def __getitem__(self, row_number):
return [SimpleNamespace(value=value) for value in self.rows[row_number - 1]]
class DetectColumnTypeTests(unittest.TestCase):
def test_index_headers_are_not_misclassified_as_count(self):
for header in ("Index", "Salary Index", "Engineering Index", "Median salary"):
with self.subTest(header=header):
self.assertEqual(detect_column_type(header), "index")
def test_single_letter_n_only_matches_as_a_token(self):
self.assertEqual(detect_column_type("Employee n"), "count")
self.assertEqual(detect_column_type("Engineering"), None)
def test_count_headers_still_match_common_labels(self):
for header in ("Count", "Engineering Count", "Antal medarbejdere"):
with self.subTest(header=header):
self.assertEqual(detect_column_type(header), "count")
def test_count_inside_word_does_not_make_count_header(self):
self.assertIsNone(detect_column_type("Accounting Total"))
self.assertEqual(detect_column_type("Accounting Index"), "index")
def test_danish_compound_headers_still_match(self):
self.assertEqual(detect_column_type("Lønindeks"), "index")
def test_compound_patterns_match_as_substring_but_others_do_not(self):
# A compound token (Danish "løn") matches inside a glued header word.
self.assertTrue(header_matches("lønindeks", INDEX_PATTERNS))
# A pattern that is not a compound token ("salary") only matches as a
# whole token, so it must not match inside an unrelated glued word.
self.assertFalse(header_matches("salaryindex", INDEX_PATTERNS))
self.assertTrue(header_matches("salary index", INDEX_PATTERNS))
def test_parse_sheet_preserves_category_name_with_letter_n(self):
ws = FakeWorksheet([
("Company", "Engineering Count", "Engineering Index"),
("Example Corp", 12, 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"]["engineering"], {"count": 12, "index": 105.5})
def test_parse_sheet_groups_accounting_count_index_pair(self):
ws = FakeWorksheet([
("Company", "Accounting Count", "Accounting Index"),
("Example Corp", 12, 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"]["accounting"], {"count": 12, "index": 105.5})
def test_parse_sheet_normalizes_paired_category_name_with_underscores(self):
ws = FakeWorksheet([
("Company", "Software Engineering Count", "Software Engineering Index"),
("Example Corp", 8, 110.0),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"]["software_engineering"], {"count": 8, "index": 110.0})
def test_parse_sheet_detects_company_column_with_token_header(self):
# Real-world salary sheets rarely use the bare token "Company";
# headers like "Company Name" / "Employer Name" must still be
# detected as the company column (previously silently skipped -> []).
for header in ("Company", "Company Name", "Employer Name"):
with self.subTest(header=header):
ws = FakeWorksheet([
(header, "Salary"),
("Example Corp", 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["company"], "Example Corp")
self.assertEqual(
companies[0]["categories"]["salary"], {"index": 105.5}
)
def test_parse_sheet_detects_city_column_with_token_header(self):
# City headers are matched with the same token-based header_matches()
# used for the company column, not exact string equality. Real-world
# sheets rarely use the bare token "City" or "Kommune" alone; headers
# like "City Name" / "City/Kommune" must still be detected as the city
# column (previously silently left as city_col=None -> empty city).
for header in ("City", "City Name", "Kommune", "City/Kommune"):
with self.subTest(header=header):
ws = FakeWorksheet([
("Company", header, "Salary"),
("Example Corp", "Aarhus", 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["city"], "Aarhus")
def test_parse_sheet_handles_ragged_rows(self):
# openpyxl's read_only mode yields ragged tuples for dimension-less
# workbooks: a row can be shorter than the header. A company row that
# omits its city and category cells must parse without an IndexError,
# be retained, and get an empty city.
ws = FakeWorksheet([
("Company", "City", "Engineering Count", "Engineering Index"),
("Example Corp",),
("Other Corp", "Aarhus", 12, 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 2)
self.assertEqual(companies[0]["company"], "Example Corp")
self.assertEqual(companies[0]["city"], "")
self.assertEqual(companies[0]["categories"], {})
self.assertEqual(companies[1]["categories"]["engineering"], {"count": 12, "index": 105.5})
def test_parse_sheet_skips_row_shorter_than_company_column(self):
# A ragged row that ends before the company column has no company cell
# at all; it must be skipped, not crash the parse.
ws = FakeWorksheet([
("Notes", "Company", "Salary Index"),
("stray",),
("", "Example Corp", 105.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["company"], "Example Corp")
def test_skips_free_text_column(self):
# A free-text "Notes" column must not become a bogus salary category.
ws = FakeWorksheet([
("Company", "Salary Index", "Notes"),
("Example Corp", 105.5, "good"),
])
companies = parse_sheet(ws)
self.assertIn("salary_index", companies[0]["categories"])
self.assertNotIn("notes", companies[0]["categories"])
def test_skips_numeric_identifier_column(self):
# A numeric "Id" column (employee id) must not be treated as a salary index.
ws = FakeWorksheet([
("Company", "Salary Index", "Id"),
("Example Corp", 105.5, 7),
])
companies = parse_sheet(ws)
self.assertIn("salary_index", companies[0]["categories"])
self.assertNotIn("id", companies[0]["categories"])
def test_keeps_numeric_salary_column(self):
# A genuine numeric salary column still produces a salary category.
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", 105.5),
])
companies = parse_sheet(ws)
self.assertIn("salary_index", companies[0]["categories"])
self.assertEqual(companies[0]["categories"]["salary_index"], {"index": 105.5})
def test_parse_sheet_accepts_comma_decimal_string_values(self):
# Locale-formatted Excel exports can carry numeric cells as strings.
# Danish decimal commas must not be silently dropped by float().
ws = FakeWorksheet([
("Company", "Engineering Count", "Engineering Index"),
("Example Corp", "12,0", "108,5"),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"]["engineering"],
{"count": 12, "index": 108.5},
)
def test_parse_sheet_accepts_danish_thousands_and_decimal_string(self):
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", "1.234,5"),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"]["salary_index"],
{"index": 1234.5},
)
def test_parse_sheet_skips_ambiguous_single_comma_thousands_string(self):
# In an English-locale export, "1,234" is probably 1234, but in a
# decimal-comma locale it could be 1.234. Preserve the old safe-skip
# behavior instead of guessing and writing a 1000x-wrong salary value.
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", "1,234"),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"], {})
def test_parse_sheet_skips_ambiguous_single_dot_thousands_string(self):
# "1.234" is the dot-side mirror of the comma guard above: in a
# decimal-dot locale it is 1.234, while a Danish export (whole
# thousands, no decimal comma, e.g. "60.000") means 1234/60000.
# float() used to write the 1000x-smaller value silently - the
# same never-guess policy must apply to both separators.
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", "1.234"),
])
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"], {})
def test_parse_sheet_pairs_interleaved_count_index_columns_by_name(self):
ws = FakeWorksheet([
("Company", "Antal kvinder", "Antal mænd", "Kvinder indeks", "Mænd indeks"),
("Example Corp", 15, 20, 95.0, 108.0),
])
companies = parse_sheet(ws)
categories = companies[0]["categories"]
self.assertEqual(categories["kvinder"], {"count": 15, "index": 95.0})
self.assertEqual(categories["mænd"], {"count": 20, "index": 108.0})
def test_standalone_count_column_is_stored_as_count_not_index(self):
# A count column with no matching index column (e.g. a lone total
# headcount) is still count data. It must not be emitted as a salary
# index, which salary_lookup would render with a bogus "vs baseline"
# percentage. The paired category alongside it is unaffected.
ws = FakeWorksheet([
("Company", "Antal", "IT Count", "IT Index"),
("Example Corp", 250, 30, 108.5),
])
companies = parse_sheet(ws)
categories = companies[0]["categories"]
self.assertEqual(categories["antal"], {"count": 250})
self.assertEqual(categories["it"], {"count": 30, "index": 108.5})
def test_parse_sheet_non_adjacent_columns_no_cross_match(self):
ws = FakeWorksheet([
("Company", "Count_A", "Count_B", "Index_A", "Index_B"),
("Example Corp", 10, 20, 100.0, 200.0),
])
companies = parse_sheet(ws)
categories = companies[0]["categories"]
self.assertEqual(categories["a"], {"count": 10, "index": 100.0})
self.assertEqual(categories["b"], {"count": 20, "index": 200.0})
def test_parse_sheet_ignores_citation_row_mentioning_company_pattern_word(self):
# A title/source-citation row above the real header - standard in
# real Danish union/statistics exports - can contain a stray
# company-pattern word ("arbejdsgiver" = employer) in running prose.
# It must not be mistaken for the header: that misreads the real
# header row as data (producing a bogus "Firma" company) and drops
# every real company's salary data (issue #414).
ws = FakeWorksheet([
("Lønstatistik 2025",),
("Kilde: Medlemsundersøgelse opdelt efter arbejdsgiver og branche",),
(),
("Firma", "By", "Antal alle", "Lønindeks alle"),
("Novo Nordisk A/S", "Bagsværd", 500, 108.5),
("Ørsted A/S", "Fredericia", 200, 105.2),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 2)
self.assertEqual(companies[0]["company"], "Novo Nordisk A/S")
self.assertEqual(companies[0]["city"], "Bagsværd")
self.assertEqual(companies[0]["categories"]["alle"], {"count": 500, "index": 108.5})
self.assertEqual(companies[1]["company"], "Ørsted A/S")
def test_parse_sheet_rejects_citation_row_with_count_word_in_same_cell(self):
# Corroboration must come from a DIFFERENT cell than the company
# match. A single free-text sentence can pack both a company-pattern
# word and a count-pattern word together (e.g. "... opdelt efter
# arbejdsgiver, antal svar 1234") - same-cell corroboration must not
# be enough, or this citation row reintroduces the bogus-header bug.
ws = FakeWorksheet([
("Lønstatistik 2025",),
("Kilde: undersøgelse opdelt efter arbejdsgiver, antal svar 1234",),
(),
("Firma", "By", "Antal alle", "Lønindeks alle"),
("Novo Nordisk A/S", "Bagsværd", 500, 108.5),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["company"], "Novo Nordisk A/S")
self.assertEqual(companies[0]["categories"]["alle"], {"count": 500, "index": 108.5})
def test_parse_sheet_falls_back_when_no_row_has_cross_cell_corroboration(self):
# A header with only untyped salary columns (no header matches a
# known city/count/index pattern - "Base pay"/"Bonus" don't) has
# nothing to corroborate against in any row. The strict cross-cell
# check must fall back to the original any-cell-mentions-company
# rule rather than failing to find a header at all.
ws = FakeWorksheet([
("Company", "Base pay 2025", "Bonus 2025"),
("Example Corp", 55000, 5000),
])
companies = parse_sheet(ws)
self.assertEqual(len(companies), 1)
self.assertEqual(companies[0]["company"], "Example Corp")
self.assertEqual(companies[0]["categories"]["base_pay_2025"], {"index": 55000.0})
self.assertEqual(companies[0]["categories"]["bonus_2025"], {"index": 5000.0})
def test_parse_sheet_warns_when_no_salary_columns_detected(self):
# A header row with only company/city columns and no salary data
# is a strong signal something is wrong (a misdetected header row,
# or a sheet with no salary data at all) - it should be flagged,
# not silently reported as a successful conversion.
ws = FakeWorksheet([
("Company", "City"),
("Example Corp", "Aarhus"),
])
stderr = io.StringIO()
with redirect_stderr(stderr):
companies = parse_sheet(ws)
self.assertEqual(companies[0]["categories"], {})
self.assertIn("No salary data columns detected", stderr.getvalue())
if __name__ == "__main__":
unittest.main()
class ParseNumericCellLocaleTests(unittest.TestCase):
# The separator that appears LAST is the decimal separator. Assuming
# European ("." thousands, "," decimal) for every both-separator string
# turned a US "1,234.56" into 1.23456 - a silent 1000x corruption that
# flowed into salary_data.json and negotiation advice.
def test_us_thousands_and_decimal_string(self):
self.assertEqual(parse_numeric_cell("1,234.56"), 1234.56)
def test_us_multiple_thousands_groups(self):
self.assertEqual(parse_numeric_cell("1,234,567.89"), 1234567.89)
def test_european_thousands_and_decimal_string(self):
self.assertEqual(parse_numeric_cell("1.234,56"), 1234.56)
def test_european_multiple_thousands_groups(self):
self.assertEqual(parse_numeric_cell("1.234.567,89"), 1234567.89)
class BareCountIndexPairingTests(unittest.TestCase):
"""A count/index pair whose headers carry no category word is still a pair.
"Count" + "Index" (Danish "Antal" + "Lønindeks") both strip to an empty
category name, and the pairing loop used to require a non-empty name on
both sides, so the single-category layout the README describes as
"auto-pairs count/index columns" came out as two unrelated standalone
columns. salary_lookup then rendered the count row with "N/A*" for the
index - "too few employees to publish (privacy)" - about a company whose
headcount was right there in the file. The literal name is asserted (not
the module constant) so the cases run, and fail, against the old converter.
"""
DEFAULT_CATEGORY = "all_employees"
def test_bare_english_pair_is_paired_under_the_default_category(self):
ws = FakeWorksheet([
("Company", "City", "Count", "Index"),
("Acme Corp", "Copenhagen", 500, 108.5),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"],
{self.DEFAULT_CATEGORY: {"count": 500, "index": 108.5}},
)
def test_bare_danish_pair_is_paired_under_the_default_category(self):
ws = FakeWorksheet([
("Firma", "By", "Antal", "Lønindeks"),
("Acme Corp", "Aarhus", 500, 108.5),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"],
{self.DEFAULT_CATEGORY: {"count": 500, "index": 108.5}},
)
def test_bare_pair_does_not_cross_pair_with_a_named_category(self):
# The bare pair and the named pair coexist; neither steals the other's
# column, and a lone "Antal" with no bare index column stays standalone
# (pinned separately by test_standalone_count_column_is_stored_as_count_not_index).
ws = FakeWorksheet([
("Company", "Antal", "IT Count", "IT Index", "Lønindeks"),
("Acme Corp", 500, 30, 112.0, 108.5),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"],
{
self.DEFAULT_CATEGORY: {"count": 500, "index": 108.5},
"it": {"count": 30, "index": 112.0},
},
)
def test_lookup_renders_the_bare_pair_as_one_row_without_the_privacy_footnote_firing(self):
# End to end through the documented path: converter output is what
# salary_lookup.format_entry displays during /apply. Before the fix the
# same sheet produced a "Count 500 N/A*" row plus an "Index - 108.5"
# row - the N/A* asserting a privacy suppression that never happened.
ws = FakeWorksheet([
("Company", "City", "Count", "Index"),
("Acme Corp", "Copenhagen", 500, 108.5),
])
entry = parse_sheet(ws)[0]
rendered = format_entry(entry, {"index_baseline": 100, "index_label": "Index"})
self.assertNotIn("N/A*", rendered.split("* N/A =")[0])
self.assertRegex(rendered, r"All Employees\s+500\s+108\.5\s+\+8\.5%")
self.assertNotRegex(rendered, r"^\s*Count\s+500", )
class CompoundCategoryPairingTests(unittest.TestCase):
def test_parse_sheet_pairs_danish_compound_index_with_count(self):
# "Lønindeks alle" is *detected* as an index column via
# COMPOUND_PATTERNS, but the derived category name must also lose the
# compound word or it can never pair with "Antal alle" ("alle" vs
# "lønindeks alle") - exactly the locale the compound support exists for.
ws = FakeWorksheet([
("Firma", "Antal alle", "Lønindeks alle"),
("Example Corp", 12, 118.0),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"]["alle"],
{"count": 12, "index": 118.0},
)
def test_parse_sheet_sheet_level_us_locale_value(self):
ws = FakeWorksheet([
("Company", "Salary Index"),
("Example Corp", "1,234.56"),
])
companies = parse_sheet(ws)
self.assertEqual(
companies[0]["categories"]["salary_index"],
{"index": 1234.56},
)