mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 16:46:24 +00:00
Initial release: AI-powered job application framework
Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
committed by
Mads Lorentzen
co-authored by
Claude Opus 4.6
commit
c66d599d75
@@ -0,0 +1,197 @@
|
||||
export const BASE_URL = "https://www.jobindex.dk"
|
||||
|
||||
export function writeError(error: string, code: string): void {
|
||||
process.stderr.write(JSON.stringify({ error, code }) + "\n")
|
||||
}
|
||||
|
||||
export async function apiFetch<T>(path: string, params?: Record<string, string>): Promise<T> {
|
||||
let url = `${BASE_URL}${path}`
|
||||
if (params && Object.keys(params).length > 0) {
|
||||
const qs = new URLSearchParams(params)
|
||||
url += `?${qs.toString()}`
|
||||
}
|
||||
|
||||
const maxRetries = 6
|
||||
let delay = 500
|
||||
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
||||
const response = await fetch(url)
|
||||
if (response.status === 429 || response.status >= 500) {
|
||||
if (attempt === maxRetries) {
|
||||
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
||||
}
|
||||
const jitter = Math.floor(Math.random() * 500)
|
||||
await new Promise((resolve) => setTimeout(resolve, delay + jitter))
|
||||
delay = Math.min(delay * 2, 5000)
|
||||
continue
|
||||
}
|
||||
if (!response.ok) {
|
||||
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
||||
}
|
||||
return response.json() as Promise<T>
|
||||
}
|
||||
throw new Error("API request failed after max retries")
|
||||
}
|
||||
|
||||
export async function htmlFetch(url: string): Promise<string> {
|
||||
const maxRetries = 6
|
||||
let delay = 500
|
||||
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
||||
const response = await fetch(url, {
|
||||
headers: {
|
||||
"User-Agent": "Mozilla/5.0 (compatible; jobindex-cli/1.0)",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "da,en;q=0.9",
|
||||
},
|
||||
redirect: "follow",
|
||||
})
|
||||
if (response.status === 429 || response.status >= 500) {
|
||||
if (attempt === maxRetries) {
|
||||
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
||||
}
|
||||
const jitter = Math.floor(Math.random() * 500)
|
||||
await new Promise((resolve) => setTimeout(resolve, delay + jitter))
|
||||
delay = Math.min(delay * 2, 5000)
|
||||
continue
|
||||
}
|
||||
if (response.status === 404) {
|
||||
throw new Error(`Job not found`)
|
||||
}
|
||||
if (!response.ok) {
|
||||
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
||||
}
|
||||
return response.text()
|
||||
}
|
||||
throw new Error("Request failed after max retries")
|
||||
}
|
||||
|
||||
export interface JobCard {
|
||||
id: string
|
||||
title: string
|
||||
company: string | null
|
||||
companyUrl: string | null
|
||||
location: string | null
|
||||
date: string | null
|
||||
url: string
|
||||
description: string | null
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode HTML entities in text
|
||||
*/
|
||||
function decodeHtmlEntities(text: string): string {
|
||||
return text
|
||||
.replace(/&/g, "&")
|
||||
.replace(/</g, "<")
|
||||
.replace(/>/g, ">")
|
||||
.replace(/"/g, '"')
|
||||
.replace(/'/g, "'")
|
||||
.replace(/'/g, "'")
|
||||
.replace(/&#(\d+);/g, (_, code) => String.fromCharCode(parseInt(code, 10)))
|
||||
.replace(/ /g, " ")
|
||||
}
|
||||
|
||||
/**
|
||||
* Strip HTML tags from text
|
||||
*/
|
||||
function stripTags(html: string): string {
|
||||
return html.replace(/<[^>]+>/g, "").trim()
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse job cards from result_list_box_html using regex.
|
||||
* node-html-parser has nesting bugs with this specific HTML structure
|
||||
* (unclosed tags inside buttons cause incorrect DOM tree).
|
||||
* Regex parsing is more reliable for this specific HTML format.
|
||||
*/
|
||||
export function parseJobCards(html: string): JobCard[] {
|
||||
const results: JobCard[] = []
|
||||
|
||||
// Split HTML by jobad-wrapper to get individual card HTML chunks
|
||||
const wrapperPattern = /<div[^>]+id="jobad-wrapper-(h\d+|r\d+)"[^>]*>([\s\S]*?)(?=<div[^>]+id="jobad-wrapper-|$)/g
|
||||
|
||||
let match: RegExpExecArray | null
|
||||
while ((match = wrapperPattern.exec(html)) !== null) {
|
||||
const id = match[1]
|
||||
const cardHtml = match[2]
|
||||
|
||||
// Extract title: look for <h4>...<a|A href="...">Title</a>...</h4>
|
||||
const titleMatch = cardHtml.match(/<h4[^>]*>[\s\S]*?<[Aa][^>]+href="([^"]+)"[^>]*>([\s\S]*?)<\/[Aa]>/i)
|
||||
if (!titleMatch) continue
|
||||
const rawTitle = stripTags(titleMatch[2])
|
||||
const title = decodeHtmlEntities(rawTitle)
|
||||
if (!title) continue
|
||||
|
||||
// Determine URL: prefer jobindex.dk /jobannonce/ URL, fallback to constructed URL
|
||||
let url: string
|
||||
const jobannonce = cardHtml.match(/href="(https:\/\/www\.jobindex\.dk\/jobannonce\/[^"]+)"/)
|
||||
if (jobannonce) {
|
||||
url = jobannonce[1]
|
||||
} else {
|
||||
// Construct canonical URL from ID
|
||||
url = `${BASE_URL}/jobannonce/${id}`
|
||||
}
|
||||
|
||||
// Extract company: <a ...> inside jix-toolbar-top__company
|
||||
let company: string | null = null
|
||||
let companyUrl: string | null = null
|
||||
const companySection = cardHtml.match(/class="jix-toolbar-top__company"[^>]*>([\s\S]*?)<\/div>/i)
|
||||
if (companySection) {
|
||||
const companyLinkMatch = companySection[1].match(/<[Aa][^>]+href="([^"]+)"[^>]*>([\s\S]*?)<\/[Aa]>/i)
|
||||
if (companyLinkMatch) {
|
||||
company = decodeHtmlEntities(stripTags(companyLinkMatch[2])) || null
|
||||
companyUrl = companyLinkMatch[1] || null
|
||||
}
|
||||
}
|
||||
|
||||
// Extract location: <span class="jix_robotjob--area">Location</span>
|
||||
const locMatch = cardHtml.match(/<span[^>]+class="jix_robotjob--area"[^>]*>([\s\S]*?)<\/span>/i)
|
||||
const location = locMatch ? decodeHtmlEntities(stripTags(locMatch[1])) || null : null
|
||||
|
||||
// Extract date: <time datetime="YYYY-MM-DD">
|
||||
const dateMatch = cardHtml.match(/<time[^>]+datetime="([^"]+)"/)
|
||||
const date = dateMatch ? dateMatch[1] : null
|
||||
|
||||
// Extract description: first <p class="..."> or first standalone <p> (not in toolbar)
|
||||
// Skip the toolbar/menu section and look for the description paragraph
|
||||
let description: string | null = null
|
||||
const innerSection = cardHtml.match(/class="PaidJob-inner"[^>]*>([\s\S]*?)(?:<\/div>\s*<\/div>|$)/i) ||
|
||||
cardHtml.match(/class="jix_robotjob-inner"[^>]*>([\s\S]*?)(?:<\/div>\s*<\/div>|$)/i)
|
||||
if (innerSection) {
|
||||
const pMatch = innerSection[1].match(/<p[^>]*>([\s\S]*?)<\/p>/i)
|
||||
if (pMatch) {
|
||||
const text = decodeHtmlEntities(stripTags(pMatch[1]))
|
||||
description = text.length > 0 ? text.substring(0, 300) : null
|
||||
}
|
||||
} else {
|
||||
// Fallback: look for p after the jobannonce link
|
||||
const pMatches = [...cardHtml.matchAll(/<p[^>]*>([\s\S]*?)<\/p>/gi)]
|
||||
for (const pm of pMatches) {
|
||||
const text = decodeHtmlEntities(stripTags(pm[1]))
|
||||
if (text.length > 20) {
|
||||
description = text.substring(0, 300)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
results.push({
|
||||
id,
|
||||
title,
|
||||
company: company || null,
|
||||
companyUrl: companyUrl || null,
|
||||
location: location || null,
|
||||
date: date || null,
|
||||
url,
|
||||
description: description || null,
|
||||
})
|
||||
}
|
||||
|
||||
return results
|
||||
}
|
||||
|
||||
export function parseHitCount(html: string): number {
|
||||
const match = html.match(/af <strong>([\d.]+)<\/strong>/)
|
||||
if (!match) return 0
|
||||
const numStr = match[1].replace(/\./g, "")
|
||||
return parseInt(numStr, 10) || 0
|
||||
}
|
||||
Reference in New Issue
Block a user