mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 00:26:26 +00:00
refactor(salary): optimize search match scoring and normalize Excel category keys (#101)
This commit improves the performance and consistency of the salary tools: - Redundant query normalization and word extraction are eliminated in salary_lookup.py by pre-calculating representations once before the search loop. - A match_score_optimized helper is introduced to perform the comparison using the pre-calculated query data, preserving full backward compatibility for match_score. - Normalization in tools/convert_salary_excel.py is unified: paired column headers now consistently substitute spaces and dashes with underscores (e.g. 'software_engineering') to match the single-column formatting. - Unit test coverage is significantly expanded in tests/test_salary_lookup.py and tests/test_convert_salary_excel.py to cover normalization, anglicization, search filtering, and matching behaviors.
This commit is contained in:
@@ -77,6 +77,16 @@ class DetectColumnTypeTests(unittest.TestCase):
|
||||
|
||||
self.assertEqual(companies[0]["categories"]["accounting"], {"count": 12, "index": 105.5})
|
||||
|
||||
def test_parse_sheet_normalizes_paired_category_name_with_underscores(self):
|
||||
ws = FakeWorksheet([
|
||||
("Company", "Software Engineering Count", "Software Engineering Index"),
|
||||
("Example Corp", 8, 110.0),
|
||||
])
|
||||
|
||||
companies = parse_sheet(ws)
|
||||
|
||||
self.assertEqual(companies[0]["categories"]["software_engineering"], {"count": 8, "index": 110.0})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
Reference in New Issue
Block a user