mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 08:36:25 +00:00
refactor(salary): make compound-word matching locale-agnostic (#94)
* refactor(salary): make compound-word matching locale-agnostic The Excel column detector hardcoded a DANISH_COMPOUND_PATTERNS set inside header_matches(), so the compound-word matching that helps Danish headers (e.g. "lønindeks") was baked into the algorithm by name and unavailable to any other locale without editing the source. Rename it to COMPOUND_PATTERNS and pass it as a parameter (default unchanged, so the Danish demonstration data behaves identically). A different-locale spreadsheet can now supply its own compound tokens via header_matches(..., compound_patterns=...). Add a test covering both the preserved default and the parameterized path. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * refactor(salary): drop unused compound_patterns parameter Per review: keep the DANISH_COMPOUND_PATTERNS -> COMPOUND_PATTERNS rename (universal template naming, defaults still Danish), but remove the compound_patterns= parameter. No caller passes a custom set, and a fork adapting another locale edits the module-level constant either way, so parameterizing it is speculative generality per CONTRIBUTING.md. header_matches() now reads COMPOUND_PATTERNS directly. Test updated to verify compound-vs-whole-token matching against the constant. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
cd7c22325b
commit
a278ad7a50
@@ -1,7 +1,12 @@
|
|||||||
import unittest
|
import unittest
|
||||||
from types import SimpleNamespace
|
from types import SimpleNamespace
|
||||||
|
|
||||||
from tools.convert_salary_excel import detect_column_type, parse_sheet
|
from tools.convert_salary_excel import (
|
||||||
|
INDEX_PATTERNS,
|
||||||
|
detect_column_type,
|
||||||
|
header_matches,
|
||||||
|
parse_sheet,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class FakeWorksheet:
|
class FakeWorksheet:
|
||||||
@@ -44,6 +49,14 @@ class DetectColumnTypeTests(unittest.TestCase):
|
|||||||
def test_danish_compound_headers_still_match(self):
|
def test_danish_compound_headers_still_match(self):
|
||||||
self.assertEqual(detect_column_type("Lønindeks"), "index")
|
self.assertEqual(detect_column_type("Lønindeks"), "index")
|
||||||
|
|
||||||
|
def test_compound_patterns_match_as_substring_but_others_do_not(self):
|
||||||
|
# A compound token (Danish "løn") matches inside a glued header word.
|
||||||
|
self.assertTrue(header_matches("lønindeks", INDEX_PATTERNS))
|
||||||
|
# A pattern that is not a compound token ("salary") only matches as a
|
||||||
|
# whole token, so it must not match inside an unrelated glued word.
|
||||||
|
self.assertFalse(header_matches("salaryindex", INDEX_PATTERNS))
|
||||||
|
self.assertTrue(header_matches("salary index", INDEX_PATTERNS))
|
||||||
|
|
||||||
def test_parse_sheet_preserves_category_name_with_letter_n(self):
|
def test_parse_sheet_preserves_category_name_with_letter_n(self):
|
||||||
ws = FakeWorksheet([
|
ws = FakeWorksheet([
|
||||||
("Company", "Engineering Count", "Engineering Index"),
|
("Company", "Engineering Count", "Engineering Index"),
|
||||||
|
|||||||
@@ -42,18 +42,28 @@ COMPANY_PATTERNS = {"firma", "company", "virksomhed", "employer", "arbejdsgiver"
|
|||||||
CITY_PATTERNS = {"by", "city", "kommune", "location", "lokation", "sted"}
|
CITY_PATTERNS = {"by", "city", "kommune", "location", "lokation", "sted"}
|
||||||
COUNT_PATTERNS = {"antal", "count", "number", "n", "employees", "medarbejdere"}
|
COUNT_PATTERNS = {"antal", "count", "number", "n", "employees", "medarbejdere"}
|
||||||
INDEX_PATTERNS = {"indeks", "index", "idx", "salary", "løn", "median", "average", "gennemsnit"}
|
INDEX_PATTERNS = {"indeks", "index", "idx", "salary", "løn", "median", "average", "gennemsnit"}
|
||||||
DANISH_COMPOUND_PATTERNS = {"antal", "indeks", "løn", "gennemsnit", "medarbejdere"}
|
# "Compound" tokens: pattern words allowed to match as a substring of a larger
|
||||||
|
# header token, for languages that glue words together (e.g. Danish "lønindeks"
|
||||||
|
# -> løn + indeks). Languages that write headers as separate words need none.
|
||||||
|
# Ships populated for this repo's Danish demonstration data; a fork targeting
|
||||||
|
# another locale edits this constant.
|
||||||
|
COMPOUND_PATTERNS = {"antal", "indeks", "løn", "gennemsnit", "medarbejdere"}
|
||||||
|
|
||||||
|
|
||||||
def header_matches(header, patterns):
|
def header_matches(header, patterns):
|
||||||
"""Return True when a header contains a meaningful pattern match."""
|
"""Return True when a header contains a meaningful pattern match.
|
||||||
|
|
||||||
|
Patterns match whole tokens; any pattern also listed in
|
||||||
|
``COMPOUND_PATTERNS`` may additionally match as a substring, to handle
|
||||||
|
languages that form compound words.
|
||||||
|
"""
|
||||||
h = header.lower().strip()
|
h = header.lower().strip()
|
||||||
tokens = set(re.findall(r"[a-zæøåöäü0-9]+", h))
|
tokens = set(re.findall(r"[a-zæøåöäü0-9]+", h))
|
||||||
|
|
||||||
for p in patterns:
|
for p in patterns:
|
||||||
if p in tokens:
|
if p in tokens:
|
||||||
return True
|
return True
|
||||||
if p in DANISH_COMPOUND_PATTERNS and p in h:
|
if p in COMPOUND_PATTERNS and p in h:
|
||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user