mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 08:36:25 +00:00
The pairing loop required a non-empty derived category name on both
sides, but a header with no category word - "Count" + "Index", Danish
"Antal" + "Lønindeks" - strips to an empty name, so the simplest layout
the README advertises ("auto-pairs count/index columns") came out as two
unrelated standalone categories:
{"count": {"count": 500}, "index": {"index": 108.5}}
salary_lookup then rendered a "Count 500 N/A*" row above an
"Index - 108.5" row, and its footnote read the N/A* as "too few
employees to publish (privacy)" - a false statement about a company whose
headcount is in the file, shown during /apply's salary step. Any suffix
("Antal alle") made pairing work, which is why the shipped tests, all
suffixed, never saw it.
Pair on equal derived names, empty included, and give the nameless pair
the README's top-level category name (all_employees). A bare "Antal" with
no bare index column still stays a standalone count; named pairs
alongside are untouched.
Four new cases in test_convert_salary_excel.py, one of them rendering the
converter's output through salary_lookup.format_entry; all four fail
against the old pairing rule.
400 lines
16 KiB
Python
400 lines
16 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Convert salary data from Excel to JSON format.
|
|
|
|
This script converts an Excel file containing company salary data
|
|
into the JSON format expected by salary_lookup.py.
|
|
|
|
Prerequisites:
|
|
pip install openpyxl
|
|
|
|
Usage:
|
|
python tools/convert_salary_excel.py <path-to-excel-file>
|
|
python tools/convert_salary_excel.py <path-to-excel-file> --source "My Union Stats 2025"
|
|
python tools/convert_salary_excel.py <path-to-excel-file> --baseline 100 --baseline-desc "Index 100 = median salary"
|
|
|
|
The output file (salary_data.json) will be written to the repository root.
|
|
|
|
Expected Excel format:
|
|
- A header row with column names
|
|
- A "Company" or "Firma" column (required)
|
|
- An optional "City" or "By" column
|
|
- Any number of numeric data columns (salary index, count, etc.)
|
|
|
|
The script auto-detects the header row and column layout. For Excel files
|
|
with paired count/index columns per category, it groups them automatically.
|
|
"""
|
|
|
|
import json
|
|
import sys
|
|
import argparse
|
|
import re
|
|
from pathlib import Path
|
|
|
|
try:
|
|
import openpyxl
|
|
except ImportError:
|
|
openpyxl = None
|
|
|
|
|
|
# Column name patterns for auto-detection
|
|
COMPANY_PATTERNS = {"firma", "company", "virksomhed", "employer", "arbejdsgiver"}
|
|
CITY_PATTERNS = {"by", "city", "kommune", "location", "lokation", "sted"}
|
|
COUNT_PATTERNS = {"antal", "count", "number", "n", "employees", "medarbejdere"}
|
|
INDEX_PATTERNS = {"indeks", "index", "idx", "salary", "løn", "median", "average", "gennemsnit"}
|
|
# "Compound" tokens: pattern words allowed to match as a substring of a larger
|
|
# header token, for languages that glue words together (e.g. Danish "lønindeks"
|
|
# -> løn + indeks). Languages that write headers as separate words need none.
|
|
# Ships populated for this repo's Danish demonstration data; a fork targeting
|
|
# another locale edits this constant.
|
|
COMPOUND_PATTERNS = {"antal", "indeks", "løn", "gennemsnit", "medarbejdere"}
|
|
# Identifier columns (employee id, Danish "personnummer", etc.) are never salary
|
|
# data. They are dropped at classification so they are not mistaken for a salary
|
|
# category. Matched as whole tokens only, like other pattern sets.
|
|
ID_PATTERNS = {"id", "personnummer"}
|
|
# Category name for a count/index pair whose headers carry no category word at
|
|
# all ("Count" + "Index"). Matches the top-level category in README_SALARY_TOOL.md.
|
|
DEFAULT_CATEGORY = "all_employees"
|
|
|
|
|
|
def parse_numeric_cell(value):
|
|
"""Parse numeric Excel values, including localized string cells."""
|
|
if isinstance(value, (int, float)):
|
|
return float(value)
|
|
if not isinstance(value, str):
|
|
raise ValueError("not numeric")
|
|
|
|
text = value.strip().replace("\u00a0", " ").replace(" ", "")
|
|
if not text:
|
|
raise ValueError("not numeric")
|
|
if "," in text and "." in text:
|
|
# The separator that appears last is the decimal separator: European
|
|
# "1.234,56" and US "1,234.56" are both unambiguous here, unlike the
|
|
# single-separator cases below.
|
|
if text.rfind(",") > text.rfind("."):
|
|
text = text.replace(".", "").replace(",", ".")
|
|
else:
|
|
text = text.replace(",", "")
|
|
elif "," in text:
|
|
if re.fullmatch(r"[+-]?\d+,\d{3}", text):
|
|
raise ValueError("ambiguous comma separator")
|
|
text = text.replace(",", ".")
|
|
elif "." in text:
|
|
if re.fullmatch(r"[+-]?\d+\.\d{3}", text):
|
|
raise ValueError("ambiguous dot separator")
|
|
return float(text)
|
|
|
|
|
|
def header_matches(header, patterns):
|
|
"""Return True when a header contains a meaningful pattern match.
|
|
|
|
Patterns match whole tokens; any pattern also listed in
|
|
``COMPOUND_PATTERNS`` may additionally match as a substring, to handle
|
|
languages that form compound words.
|
|
"""
|
|
h = header.lower().strip()
|
|
tokens = set(re.findall(r"[a-zæøåöäü0-9]+", h))
|
|
|
|
for p in patterns:
|
|
if p in tokens:
|
|
return True
|
|
if p in COMPOUND_PATTERNS and p in h:
|
|
return True
|
|
return False
|
|
|
|
|
|
def strip_type_patterns(header, patterns):
|
|
"""Remove count/index words from a header to derive a category name.
|
|
|
|
Mirrors ``header_matches``: patterns strip as whole tokens, and any
|
|
pattern also listed in ``COMPOUND_PATTERNS`` additionally strips as a
|
|
substring - otherwise a compound header like "Lønindeks alle" keeps the
|
|
type word in its category name and can never pair with "Antal alle".
|
|
"""
|
|
name = header.lower()
|
|
for p in patterns:
|
|
name = re.sub(rf"(?<![a-zæøåöäü0-9]){re.escape(p)}(?![a-zæøåöäü0-9])", "", name)
|
|
if p in COMPOUND_PATTERNS:
|
|
name = name.replace(p, "")
|
|
return name.strip(" _-")
|
|
|
|
|
|
def detect_column_type(header):
|
|
"""Detect whether a column header refers to count or index data."""
|
|
if header_matches(header, COUNT_PATTERNS):
|
|
return "count"
|
|
if header_matches(header, INDEX_PATTERNS):
|
|
return "index"
|
|
return None
|
|
|
|
|
|
def parse_sheet(ws, sheet_label=None):
|
|
"""Parse a single worksheet into a list of company entries and detected categories."""
|
|
# Find header row. Two passes:
|
|
#
|
|
# Strict pass: a candidate row needs a company-pattern cell AND a
|
|
# DIFFERENT cell matching a city/count/index pattern. Corroboration must
|
|
# come from a separate cell - a single free-text sentence can pack both
|
|
# a company-pattern word and a count-pattern word together (e.g. "...
|
|
# opdelt efter arbejdsgiver, antal svar 1234"), and that must not read
|
|
# as a header any more than a citation mentioning just one of them does.
|
|
# A real header row always has these as separate columns.
|
|
#
|
|
# Fallback pass: some real headers have no recognizable city/count/index
|
|
# column at all (e.g. "Company | Base pay 2025 | Bonus 2025" - neither
|
|
# data header matches a known pattern, so they're picked up later as
|
|
# untyped/standalone categories). Nothing can corroborate a company match
|
|
# there, so if the strict pass finds no row in the first 10, fall back to
|
|
# the original any-cell-mentions-company rule.
|
|
rows = list(ws.iter_rows(min_row=1, max_row=10, values_only=False))
|
|
|
|
def _cell_texts(row):
|
|
return [str(cell.value).strip() for cell in row if cell.value]
|
|
|
|
header_row = None
|
|
for row_idx, row in enumerate(rows, start=1):
|
|
cell_texts = _cell_texts(row)
|
|
company_idxs = {i for i, t in enumerate(cell_texts) if header_matches(t, COMPANY_PATTERNS)}
|
|
if not company_idxs:
|
|
continue
|
|
other_idxs = {
|
|
i
|
|
for i, t in enumerate(cell_texts)
|
|
if header_matches(t, CITY_PATTERNS) or header_matches(t, COUNT_PATTERNS) or header_matches(t, INDEX_PATTERNS)
|
|
}
|
|
if other_idxs - company_idxs:
|
|
header_row = row_idx
|
|
break
|
|
|
|
if header_row is None:
|
|
for row_idx, row in enumerate(rows, start=1):
|
|
if any(header_matches(t, COMPANY_PATTERNS) for t in _cell_texts(row)):
|
|
header_row = row_idx
|
|
break
|
|
|
|
if header_row is None:
|
|
print(f"Warning: Could not find header row in sheet '{ws.title}'. Skipping.", file=sys.stderr)
|
|
return []
|
|
|
|
# Read headers
|
|
headers = []
|
|
for cell in ws[header_row]:
|
|
headers.append(str(cell.value).strip() if cell.value else "")
|
|
|
|
# Find company and city columns
|
|
company_col = None
|
|
city_col = None
|
|
for i, h in enumerate(headers):
|
|
if header_matches(h, COMPANY_PATTERNS):
|
|
company_col = i
|
|
elif header_matches(h, CITY_PATTERNS):
|
|
city_col = i
|
|
|
|
if company_col is None:
|
|
print(f"Warning: Could not find company column in sheet '{ws.title}'.", file=sys.stderr)
|
|
return []
|
|
|
|
# Identify data columns (everything that's not company/city or an identifier)
|
|
data_cols = []
|
|
for i, h in enumerate(headers):
|
|
if i == company_col or i == city_col or not h:
|
|
continue
|
|
if header_matches(h, ID_PATTERNS):
|
|
continue
|
|
data_cols.append((i, h))
|
|
|
|
# Group data columns by detected type and derive category names
|
|
count_cols = []
|
|
index_cols = []
|
|
untyped_cols = []
|
|
|
|
for col_idx, col_header in data_cols:
|
|
col_type = detect_column_type(col_header)
|
|
if col_type == "count":
|
|
cat_name = strip_type_patterns(col_header, COUNT_PATTERNS)
|
|
count_cols.append((col_idx, col_header, cat_name))
|
|
elif col_type == "index":
|
|
cat_name = strip_type_patterns(col_header, INDEX_PATTERNS)
|
|
index_cols.append((col_idx, col_header, cat_name))
|
|
else:
|
|
untyped_cols.append((col_idx, col_header))
|
|
|
|
# Pair count/index columns by matching category name. A bare "Count" /
|
|
# "Index" pair (Danish "Antal" / "Lønindeks") strips to an empty name on
|
|
# both sides - the single-category layout the README's "auto-pairs
|
|
# count/index columns" line describes. It is still one pair, so it gets
|
|
# the README's default category name instead of being emitted as two
|
|
# unrelated standalone columns: salary_lookup renders that split as a
|
|
# count row whose index reads "N/A*", i.e. "too few employees to publish
|
|
# (privacy)", about a company with a published headcount.
|
|
categories = []
|
|
used_counts = set()
|
|
used_indexes = set()
|
|
|
|
for ci, (c_idx, c_header, c_cat) in enumerate(count_cols):
|
|
for ii, (i_idx, i_header, i_cat) in enumerate(index_cols):
|
|
if ii in used_indexes:
|
|
continue
|
|
if c_cat == i_cat:
|
|
cat_name = (c_cat or DEFAULT_CATEGORY).replace(" ", "_").replace("-", "_")
|
|
categories.append({
|
|
"name": cat_name,
|
|
"count_col": c_idx,
|
|
"index_col": i_idx,
|
|
})
|
|
used_counts.add(ci)
|
|
used_indexes.add(ii)
|
|
break
|
|
|
|
# Remaining unmatched count columns become standalone. They are still count
|
|
# data, so tag them as such — otherwise a lone headcount would be emitted as
|
|
# a salary index and rendered with a meaningless "vs baseline" percentage.
|
|
for ci, (c_idx, c_header, _) in enumerate(count_cols):
|
|
if ci not in used_counts:
|
|
categories.append(
|
|
{"name": c_header.lower().replace(" ", "_"), "value_col": c_idx, "field": "count"}
|
|
)
|
|
|
|
# Remaining unmatched index columns become standalone (use original header)
|
|
for ii, (i_idx, i_header, _) in enumerate(index_cols):
|
|
if ii not in used_indexes:
|
|
categories.append({"name": i_header.lower().replace(" ", "_"), "value_col": i_idx})
|
|
|
|
# Untyped columns become standalone
|
|
for col_idx, col_header in untyped_cols:
|
|
categories.append({"name": col_header.lower().replace(" ", "_"), "value_col": col_idx})
|
|
|
|
if not categories:
|
|
print(
|
|
f"Warning: No salary data columns detected in sheet '{ws.title}' "
|
|
"(only a company/city column was found) - the header row may be "
|
|
"wrong, or this sheet has no salary data.",
|
|
file=sys.stderr,
|
|
)
|
|
|
|
# Parse data rows
|
|
companies = []
|
|
for row in ws.iter_rows(min_row=header_row + 1, values_only=True):
|
|
if company_col >= len(row) or not row[company_col]:
|
|
continue
|
|
|
|
company_name = str(row[company_col]).strip()
|
|
if city_col is not None and city_col < len(row) and row[city_col]:
|
|
city_name = str(row[city_col]).strip()
|
|
else:
|
|
city_name = ""
|
|
|
|
entry = {
|
|
"company": company_name,
|
|
"city": city_name,
|
|
"categories": {},
|
|
}
|
|
|
|
for cat in categories:
|
|
cat_name = cat["name"]
|
|
if "count_col" in cat and "index_col" in cat:
|
|
count_val = None
|
|
index_val = None
|
|
if cat["count_col"] < len(row) and row[cat["count_col"]] is not None:
|
|
try:
|
|
count_val = int(parse_numeric_cell(row[cat["count_col"]]))
|
|
except (ValueError, TypeError):
|
|
pass
|
|
if cat["index_col"] < len(row) and row[cat["index_col"]] is not None:
|
|
try:
|
|
index_val = parse_numeric_cell(row[cat["index_col"]])
|
|
except (ValueError, TypeError):
|
|
pass
|
|
# A count/index pair that is entirely empty for this row carries
|
|
# no salary information, so skip it rather than emit nulls.
|
|
if count_val is None and index_val is None:
|
|
continue
|
|
entry["categories"][cat_name] = {"count": count_val, "index": index_val}
|
|
elif "value_col" in cat:
|
|
if cat["value_col"] < len(row) and row[cat["value_col"]] is not None:
|
|
val = row[cat["value_col"]]
|
|
try:
|
|
val = parse_numeric_cell(val)
|
|
except (ValueError, TypeError):
|
|
# Non-numeric standalone value (e.g. a free-text "Notes"
|
|
# column) is not salary data; skip it for this row.
|
|
continue
|
|
field = cat.get("field", "index")
|
|
entry["categories"][cat_name] = {field: int(val) if field == "count" else val}
|
|
|
|
companies.append(entry)
|
|
|
|
return companies
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
description="Convert salary Excel data to JSON"
|
|
)
|
|
parser.add_argument("excel_file", help="Path to the Excel file with salary data")
|
|
parser.add_argument(
|
|
"--output", default=None,
|
|
help="Output JSON file path (default: salary_data.json in repo root)",
|
|
)
|
|
parser.add_argument(
|
|
"--source", default=None,
|
|
help="Name of the data source (e.g., 'Union Statistics 2025')",
|
|
)
|
|
parser.add_argument(
|
|
"--baseline", type=float, default=100,
|
|
help="Baseline value for index comparison (default: 100)",
|
|
)
|
|
parser.add_argument(
|
|
"--baseline-desc", default=None,
|
|
help="Description of what the baseline means (e.g., 'Index 100 = median salary')",
|
|
)
|
|
args = parser.parse_args()
|
|
|
|
excel_path = Path(args.excel_file)
|
|
if not excel_path.exists():
|
|
print(f"Error: File not found: {excel_path}", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
if openpyxl is None:
|
|
print("Error: openpyxl is required. Install it with: pip install openpyxl", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
output_path = Path(args.output) if args.output else Path(__file__).parent.parent / "salary_data.json"
|
|
|
|
print(f"Reading: {excel_path}")
|
|
wb = openpyxl.load_workbook(excel_path, read_only=True, data_only=True)
|
|
|
|
all_companies = []
|
|
for sheet_name in wb.sheetnames:
|
|
print(f" Parsing sheet: {sheet_name}")
|
|
ws = wb[sheet_name]
|
|
companies = parse_sheet(ws, sheet_label=sheet_name)
|
|
all_companies.extend(companies)
|
|
|
|
wb.close()
|
|
|
|
if not all_companies:
|
|
print("Error: No data could be parsed from the Excel file.", file=sys.stderr)
|
|
print("Make sure the Excel file has a header row with a 'Company'/'Firma' column.", file=sys.stderr)
|
|
sys.exit(1)
|
|
|
|
# Build output
|
|
output = {
|
|
"metadata": {
|
|
"source": args.source or excel_path.stem,
|
|
"index_baseline": args.baseline,
|
|
"index_label": "Index",
|
|
"baseline_description": args.baseline_desc or f"Index {args.baseline} = baseline",
|
|
},
|
|
"companies": all_companies,
|
|
}
|
|
|
|
with open(output_path, "w", encoding="utf-8") as f:
|
|
json.dump(output, f, ensure_ascii=False, indent=2)
|
|
|
|
print(f"\nDone! Wrote {len(all_companies)} company entries to {output_path}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|