mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 00:26:26 +00:00
fix(salary): detect city column from header token, not exact match (#201)
convert_salary_excel.py detected the city column via exact membership (h_lower in CITY_PATTERNS), so real headers like "City Name", "City/Kommune", or "Kommune <suffix>" never matched and every company was written with an empty city field. Switches to header_matches(h, CITY_PATTERNS) - the same whole-token matcher already used for the company, count, index, and ID columns. Same bug class as #151 (company column); bare "City"/"Kommune" inputs are unaffected. Regression test covers bare and suffixed headers. By @oscarbol09.
This commit is contained in:
@@ -104,6 +104,22 @@ class DetectColumnTypeTests(unittest.TestCase):
|
||||
companies[0]["categories"]["salary"], {"index": 105.5}
|
||||
)
|
||||
|
||||
def test_parse_sheet_detects_city_column_with_token_header(self):
|
||||
# City headers are matched with the same token-based header_matches()
|
||||
# used for the company column, not exact string equality. Real-world
|
||||
# sheets rarely use the bare token "City" or "Kommune" alone; headers
|
||||
# like "City Name" / "City/Kommune" must still be detected as the city
|
||||
# column (previously silently left as city_col=None -> empty city).
|
||||
for header in ("City", "City Name", "Kommune", "City/Kommune"):
|
||||
with self.subTest(header=header):
|
||||
ws = FakeWorksheet([
|
||||
("Company", header, "Salary"),
|
||||
("Example Corp", "Aarhus", 105.5),
|
||||
])
|
||||
companies = parse_sheet(ws)
|
||||
self.assertEqual(len(companies), 1)
|
||||
self.assertEqual(companies[0]["city"], "Aarhus")
|
||||
|
||||
def test_skips_free_text_column(self):
|
||||
# A free-text "Notes" column must not become a bogus salary category.
|
||||
ws = FakeWorksheet([
|
||||
|
||||
Reference in New Issue
Block a user