mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 00:26:26 +00:00
`detail` called fetch() directly rather than going through the CLI's own request wrappers, so it had none of the three things apiFetch/apiPost guarantee: no 429/5xx retry loop, a hand-inlined User-Agent that would drift from the exported USER_AGENT, and a timeout no wrapper test covered. A rate-limited detail page wrote API_ERROR and exited after ONE attempt; jobnet, jobbank, jobindex, linkedin, and freehire all retry up to six times on the same response. /scrape calls detail once per shortlisted posting, so a burst that tripped jobdanmark's limiter dropped those postings (no description, no deadline) while any other portal rode it out. Demonstrated by driving the real command handler with a stubbed 429 and instant timers: 1 fetch attempt and exit 1 before, 7 after (initial try plus six retries, the contract's schedule). Add htmlFetch to helpers.ts with the same backoff, timeout, and shared User-Agent as the JSON wrappers - 404 returns null so detail keeps its NOT_FOUND contract - and route detail through it. The retry-backoff, user-agent, and request-timeout suites now cover all three wrappers, and the new detail-backoff.test.ts exercises the handler path itself; its two retry cases fail against the bare fetch().
123 lines
4.2 KiB
TypeScript
123 lines
4.2 KiB
TypeScript
export const BASE_URL = "https://jobdanmark.dk"
|
|
export const USER_AGENT = "Mozilla/5.0 (compatible; jobdanmark-cli/1.0)"
|
|
|
|
export async function apiFetch<T>(path: string, params?: Record<string, string>): Promise<T> {
|
|
let url = `${BASE_URL}${path}`
|
|
if (params && Object.keys(params).length > 0) {
|
|
const qs = new URLSearchParams(params)
|
|
url += `?${qs.toString()}`
|
|
}
|
|
|
|
const maxRetries = 6
|
|
let delay = 500
|
|
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
|
const response = await fetch(url, {
|
|
headers: { "User-Agent": USER_AGENT },
|
|
signal: AbortSignal.timeout(15000),
|
|
})
|
|
if (response.status === 429 || response.status >= 500) {
|
|
if (attempt === maxRetries) {
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
|
}
|
|
const jitter = Math.floor(Math.random() * 500)
|
|
await new Promise((resolve) => setTimeout(resolve, delay + jitter))
|
|
delay = Math.min(delay * 2, 5000)
|
|
continue
|
|
}
|
|
if (!response.ok) {
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
|
}
|
|
return response.json() as Promise<T>
|
|
}
|
|
throw new Error("API request failed after max retries")
|
|
}
|
|
|
|
export async function apiPost<T>(path: string, body: unknown): Promise<T> {
|
|
const url = `${BASE_URL}${path}`
|
|
|
|
const maxRetries = 6
|
|
let delay = 500
|
|
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
|
const response = await fetch(url, {
|
|
method: "POST",
|
|
headers: {
|
|
"Content-Type": "application/json",
|
|
"User-Agent": USER_AGENT,
|
|
},
|
|
body: JSON.stringify(body),
|
|
signal: AbortSignal.timeout(15000),
|
|
})
|
|
if (response.status === 429 || response.status >= 500) {
|
|
if (attempt === maxRetries) {
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
|
}
|
|
const jitter = Math.floor(Math.random() * 500)
|
|
await new Promise((resolve) => setTimeout(resolve, delay + jitter))
|
|
delay = Math.min(delay * 2, 5000)
|
|
continue
|
|
}
|
|
if (!response.ok) {
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
|
}
|
|
return response.json() as Promise<T>
|
|
}
|
|
throw new Error("API request failed after max retries")
|
|
}
|
|
|
|
/**
|
|
* Fetch a rendered jobdanmark.dk page as text, with the same 429/5xx backoff,
|
|
* request timeout, and User-Agent as apiFetch/apiPost. `detail` reads HTML
|
|
* rather than the JSON API; it used to call fetch() directly with none of the
|
|
* three, so a rate-limited detail page failed on the first 429 while every
|
|
* other portal's detail command retried. Returns null on 404 so the caller
|
|
* keeps its own NOT_FOUND contract.
|
|
*/
|
|
export async function htmlFetch(url: string): Promise<string | null> {
|
|
const maxRetries = 6
|
|
let delay = 500
|
|
for (let attempt = 0; attempt <= maxRetries; attempt++) {
|
|
const response = await fetch(url, {
|
|
headers: {
|
|
"Accept": "text/html,application/xhtml+xml",
|
|
"User-Agent": USER_AGENT,
|
|
},
|
|
signal: AbortSignal.timeout(15000),
|
|
})
|
|
if (response.status === 429 || response.status >= 500) {
|
|
if (attempt === maxRetries) {
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
|
}
|
|
const jitter = Math.floor(Math.random() * 500)
|
|
await new Promise((resolve) => setTimeout(resolve, delay + jitter))
|
|
delay = Math.min(delay * 2, 5000)
|
|
continue
|
|
}
|
|
if (response.status === 404) {
|
|
return null
|
|
}
|
|
if (!response.ok) {
|
|
throw new Error(`API request failed: ${response.status} ${response.statusText}`)
|
|
}
|
|
return response.text()
|
|
}
|
|
throw new Error("API request failed after max retries")
|
|
}
|
|
|
|
export function writeError(error: string, code: string): void {
|
|
process.stderr.write(JSON.stringify({ error, code }) + "\n")
|
|
}
|
|
|
|
export function stripHtml(html: string): string {
|
|
return html.replace(/<[^>]*>/g, "").replace(/\s+/g, " ").trim()
|
|
}
|
|
|
|
export function normalizeSlug(input: string): string | null {
|
|
const trimmed = input.trim()
|
|
if (!trimmed) return null
|
|
const match = trimmed.match(/\/job\/([^/?#]+)/)
|
|
if (match) return match[1]
|
|
if (/^[a-zA-Z0-9_-]+$/.test(trimmed)) return trimmed
|
|
return null
|
|
}
|
|
|