mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 16:46:24 +00:00
198 lines
6.7 KiB
TypeScript
198 lines
6.7 KiB
TypeScript
export const BASE_URL = "https://www.jobindex.dk"
|
|||
|
|
|
||
|
|
export function writeError(error: string, code: string): void {
|
||
|
|
process.stderr.write(JSON.stringify({ error, code }) + "\n")
|
||
|
|
}
|
||
|
|
|
||
|
|
export async function apiFetch<T>(path: string, params?: Record<string, string>): Promise<T> {
|
||
|
|
let url = `${BASE_URL}${path}`
|
||
|
|
if (params && Object.keys(params).length > 0) {
|
||
|
|
const qs = new URLSearchParams(params)
|
||
|
|
url += `?${qs.toString()}`
|
||
|
|
}
|
||
|
|
|
||
|
|
const maxRetries = 6
|
||
|
|
let delay = 500
|
||
|
|
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
||
|
|
const response = await fetch(url)
|
||
|
|
if (response.status === 429 || response.status >= 500) {
|
||
|
|
if (attempt === maxRetries) {
|
||
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
||
|
|
}
|
||
|
|
const jitter = Math.floor(Math.random() * 500)
|
||
|
|
await new Promise((resolve) => setTimeout(resolve, delay + jitter))
|
||
|
|
delay = Math.min(delay * 2, 5000)
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
if (!response.ok) {
|
||
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
||
|
|
}
|
||
|
|
return response.json() as Promise<T>
|
||
|
|
}
|
||
|
|
throw new Error("API request failed after max retries")
|
||
|
|
}
|
||
|
|
|
||
|
|
export async function htmlFetch(url: string): Promise<string> {
|
||
|
|
const maxRetries = 6
|
||
|
|
let delay = 500
|
||
|
|
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
||
|
|
const response = await fetch(url, {
|
||
|
|
headers: {
|
||
|
|
"User-Agent": "Mozilla/5.0 (compatible; jobindex-cli/1.0)",
|
||
|
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||
|
|
"Accept-Language": "da,en;q=0.9",
|
||
|
|
},
|
||
|
|
redirect: "follow",
|
||
|
|
})
|
||
|
|
if (response.status === 429 || response.status >= 500) {
|
||
|
|
if (attempt === maxRetries) {
|
||
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
||
|
|
}
|
||
|
|
const jitter = Math.floor(Math.random() * 500)
|
||
|
|
await new Promise((resolve) => setTimeout(resolve, delay + jitter))
|
||
|
|
delay = Math.min(delay * 2, 5000)
|
||
|
|
continue
|
||
|
|
}
|
||
|
|
if (response.status === 404) {
|
||
|
|
throw new Error(`Job not found`)
|
||
|
|
}
|
||
|
|
if (!response.ok) {
|
||
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
||
|
|
}
|
||
|
|
return response.text()
|
||
|
|
}
|
||
|
|
throw new Error("Request failed after max retries")
|
||
|
|
}
|
||
|
|
|
||
|
|
export interface JobCard {
|
||
|
|
id: string
|
||
|
|
title: string
|
||
|
|
company: string | null
|
||
|
|
companyUrl: string | null
|
||
|
|
location: string | null
|
||
|
|
date: string | null
|
||
|
|
url: string
|
||
|
|
description: string | null
|
||
|
|
}
|
||
|
|
|
||
|
|
/**
|
||
|
|
* Decode HTML entities in text
|
||
|
|
*/
|
||
|
|
function decodeHtmlEntities(text: string): string {
|
||
|
|
return text
|
||
|
|
.replace(/&/g, "&")
|
||
|
|
.replace(/</g, "<")
|
||
|
|
.replace(/>/g, ">")
|
||
|
|
.replace(/"/g, '"')
|
||
|
|
.replace(/'/g, "'")
|
||
|
|
.replace(/'/g, "'")
|
||
|
|
.replace(/&#(\d+);/g, (_, code) => String.fromCharCode(parseInt(code, 10)))
|
||
|
|
.replace(/ /g, " ")
|
||
|
|
}
|
||
|
|
|
||
|
|
/**
|
||
|
|
* Strip HTML tags from text
|
||
|
|
*/
|
||
|
|
function stripTags(html: string): string {
|
||
|
|
return html.replace(/<[^>]+>/g, "").trim()
|
||
|
|
}
|
||
|
|
|
||
|
|
/**
|
||
|
|
* Parse job cards from result_list_box_html using regex.
|
||
|
|
* node-html-parser has nesting bugs with this specific HTML structure
|
||
|
|
* (unclosed tags inside buttons cause incorrect DOM tree).
|
||
|
|
* Regex parsing is more reliable for this specific HTML format.
|
||
|
|
*/
|
||
|
|
export function parseJobCards(html: string): JobCard[] {
|
||
|
|
const results: JobCard[] = []
|
||
|
|
|
||
|
|
// Split HTML by jobad-wrapper to get individual card HTML chunks
|
||
|
|
const wrapperPattern = /<div[^>]+id="jobad-wrapper-(h\d+|r\d+)"[^>]*>([\s\S]*?)(?=<div[^>]+id="jobad-wrapper-|$)/g
|
||
|
|
|
||
|
|
let match: RegExpExecArray | null
|
||
|
|
while ((match = wrapperPattern.exec(html)) !== null) {
|
||
|
|
const id = match[1]
|
||
|
|
const cardHtml = match[2]
|
||
|
|
|
||
|
|
// Extract title: look for <h4>...<a|A href="...">Title</a>...</h4>
|
||
|
|
const titleMatch = cardHtml.match(/<h4[^>]*>[\s\S]*?<[Aa][^>]+href="([^"]+)"[^>]*>([\s\S]*?)<\/[Aa]>/i)
|
||
|
|
if (!titleMatch) continue
|
||
|
|
const rawTitle = stripTags(titleMatch[2])
|
||
|
|
const title = decodeHtmlEntities(rawTitle)
|
||
|
|
if (!title) continue
|
||
|
|
|
||
|
|
// Determine URL: prefer jobindex.dk /jobannonce/ URL, fallback to constructed URL
|
||
|
|
let url: string
|
||
|
|
const jobannonce = cardHtml.match(/href="(https:\/\/www\.jobindex\.dk\/jobannonce\/[^"]+)"/)
|
||
|
|
if (jobannonce) {
|
||
|
|
url = jobannonce[1]
|
||
|
|
} else {
|
||
|
|
// Construct canonical URL from ID
|
||
|
|
url = `${BASE_URL}/jobannonce/${id}`
|
||
|
|
}
|
||
|
|
|
||
|
|
// Extract company: <a ...> inside jix-toolbar-top__company
|
||
|
|
let company: string | null = null
|
||
|
|
let companyUrl: string | null = null
|
||
|
|
const companySection = cardHtml.match(/class="jix-toolbar-top__company"[^>]*>([\s\S]*?)<\/div>/i)
|
||
|
|
if (companySection) {
|
||
|
|
const companyLinkMatch = companySection[1].match(/<[Aa][^>]+href="([^"]+)"[^>]*>([\s\S]*?)<\/[Aa]>/i)
|
||
|
|
if (companyLinkMatch) {
|
||
|
|
company = decodeHtmlEntities(stripTags(companyLinkMatch[2])) || null
|
||
|
|
companyUrl = companyLinkMatch[1] || null
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
// Extract location: <span class="jix_robotjob--area">Location</span>
|
||
|
|
const locMatch = cardHtml.match(/<span[^>]+class="jix_robotjob--area"[^>]*>([\s\S]*?)<\/span>/i)
|
||
|
|
const location = locMatch ? decodeHtmlEntities(stripTags(locMatch[1])) || null : null
|
||
|
|
|
||
|
|
// Extract date: <time datetime="YYYY-MM-DD">
|
||
|
|
const dateMatch = cardHtml.match(/<time[^>]+datetime="([^"]+)"/)
|
||
|
|
const date = dateMatch ? dateMatch[1] : null
|
||
|
|
|
||
|
|
// Extract description: first <p class="..."> or first standalone <p> (not in toolbar)
|
||
|
|
// Skip the toolbar/menu section and look for the description paragraph
|
||
|
|
let description: string | null = null
|
||
|
|
const innerSection = cardHtml.match(/class="PaidJob-inner"[^>]*>([\s\S]*?)(?:<\/div>\s*<\/div>|$)/i) ||
|
||
|
|
cardHtml.match(/class="jix_robotjob-inner"[^>]*>([\s\S]*?)(?:<\/div>\s*<\/div>|$)/i)
|
||
|
|
if (innerSection) {
|
||
|
|
const pMatch = innerSection[1].match(/<p[^>]*>([\s\S]*?)<\/p>/i)
|
||
|
|
if (pMatch) {
|
||
|
|
const text = decodeHtmlEntities(stripTags(pMatch[1]))
|
||
|
|
description = text.length > 0 ? text.substring(0, 300) : null
|
||
|
|
}
|
||
|
|
} else {
|
||
|
|
// Fallback: look for p after the jobannonce link
|
||
|
|
const pMatches = [...cardHtml.matchAll(/<p[^>]*>([\s\S]*?)<\/p>/gi)]
|
||
|
|
for (const pm of pMatches) {
|
||
|
|
const text = decodeHtmlEntities(stripTags(pm[1]))
|
||
|
|
if (text.length > 20) {
|
||
|
|
description = text.substring(0, 300)
|
||
|
|
break
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
|
||
|
|
results.push({
|
||
|
|
id,
|
||
|
|
title,
|
||
|
|
company: company || null,
|
||
|
|
companyUrl: companyUrl || null,
|
||
|
|
location: location || null,
|
||
|
|
date: date || null,
|
||
|
|
url,
|
||
|
|
description: description || null,
|
||
|
|
})
|
||
|
|
}
|
||
|
|
|
||
|
|
return results
|
||
|
|
}
|
||
|
|
|
||
|
|
export function parseHitCount(html: string): number {
|
||
|
|
const match = html.match(/af <strong>([\d.]+)<\/strong>/)
|
||
|
|
if (!match) return 0
|
||
|
|
const numStr = match[1].replace(/\./g, "")
|
||
|
|
return parseInt(numStr, 10) || 0
|
||
|
|
}
|