mirror of
https://github.com/prdlk/cv.git
synced 2026-08-02 17:31:41 +00:00
164 lines
5.5 KiB
TypeScript
164 lines
5.5 KiB
TypeScript
import type { JobExtract } from "../shared/types";
|
||||
|
|
|
|||
|
|
const UA =
|
|||
|
|
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36";
|
|||
|
|
const MAX_BYTES = 4_000_000;
|
|||
|
|
const MAX_TEXT = 24_000;
|
|||
|
|
|
|||
|
|
/** Fetch a job posting URL and extract readable text, preferring JobPosting JSON-LD. */
|
|||
|
|
export async function extractJob(url: string): Promise<JobExtract> {
|
|||
|
|
const res = await fetch(url, {
|
|||
|
|
redirect: "follow",
|
|||
|
|
headers: {
|
|||
|
|
"User-Agent": UA,
|
|||
|
|
Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|||
|
|
"Accept-Language": "en-US,en;q=0.9",
|
|||
|
|
},
|
|||
|
|
});
|
|||
|
|
if (!res.ok) throw new Error(`upstream responded ${res.status} ${res.statusText}`);
|
|||
|
|
|
|||
|
|
const raw = await readCapped(res);
|
|||
|
|
const contentType = res.headers.get("content-type") ?? "";
|
|||
|
|
if (!contentType.includes("html")) {
|
|||
|
|
return { url, title: url, text: tidy(raw).slice(0, MAX_TEXT), source: "plain" };
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
const posting = findJobPostingLd(raw);
|
|||
|
|
if (posting) return { url, title: posting.title, text: posting.text.slice(0, MAX_TEXT), source: "json-ld" };
|
|||
|
|
|
|||
|
|
const titleMatch = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(raw);
|
|||
|
|
const title = titleMatch ? tidy(decodeEntities(titleMatch[1])) : url;
|
|||
|
|
return { url, title: title || url, text: stripHtml(raw).slice(0, MAX_TEXT), source: "html" };
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
/** Stream the body with a byte cap so an unbounded response cannot exhaust memory. */
|
|||
|
|
async function readCapped(res: Response): Promise<string> {
|
|||
|
|
if (!res.body) return "";
|
|||
|
|
const reader = res.body.getReader();
|
|||
|
|
const chunks: Uint8Array[] = [];
|
|||
|
|
let bytes = 0;
|
|||
|
|
for (;;) {
|
|||
|
|
const { done, value } = await reader.read();
|
|||
|
|
if (done) break;
|
|||
|
|
bytes += value.length;
|
|||
|
|
chunks.push(value);
|
|||
|
|
if (bytes >= MAX_BYTES) {
|
|||
|
|
await reader.cancel();
|
|||
|
|
break;
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
const merged = new Uint8Array(Math.min(bytes, MAX_BYTES));
|
|||
|
|
let offset = 0;
|
|||
|
|
for (const chunk of chunks) {
|
|||
|
|
const slice = chunk.subarray(0, Math.min(chunk.length, merged.length - offset));
|
|||
|
|
merged.set(slice, offset);
|
|||
|
|
offset += slice.length;
|
|||
|
|
if (offset >= merged.length) break;
|
|||
|
|
}
|
|||
|
|
return new TextDecoder("utf-8", { fatal: false, ignoreBOM: false }).decode(merged);
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
/** Walk every <script type="application/ld+json"> looking for a schema.org JobPosting. */
|
|||
|
|
function findJobPostingLd(html: string): { title: string; text: string } | null {
|
|||
|
|
const scriptRe = /<script[^>]*type\s*=\s*["']application\/ld\+json["'][^>]*>([\s\S]*?)<\/script>/gi;
|
|||
|
|
for (let m = scriptRe.exec(html); m; m = scriptRe.exec(html)) {
|
|||
|
|
let parsed: unknown;
|
|||
|
|
try {
|
|||
|
|
parsed = JSON.parse(m[1]);
|
|||
|
|
} catch {
|
|||
|
|
continue;
|
|||
|
|
}
|
|||
|
|
const posting = findPostingNode(parsed);
|
|||
|
|
if (!posting) continue;
|
|||
|
|
|
|||
|
|
const parts: string[] = [];
|
|||
|
|
const title = typeof posting.title === "string" ? tidy(posting.title) : "";
|
|||
|
|
if (title) parts.push(title);
|
|||
|
|
if (posting.hiringOrganization && typeof posting.hiringOrganization === "object") {
|
|||
|
|
const org = posting.hiringOrganization;
|
|||
|
|
if ("name" in org && typeof org.name === "string") parts.push(`Company: ${org.name}`);
|
|||
|
|
}
|
|||
|
|
if (typeof posting.description === "string") parts.push(stripHtml(posting.description));
|
|||
|
|
if (parts.length > (title ? 1 : 0)) return { title: title || "Job Posting", text: parts.join("\n\n") };
|
|||
|
|
}
|
|||
|
|
return null;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
interface JobPostingNode {
|
|||
|
|
title?: unknown;
|
|||
|
|
description?: unknown;
|
|||
|
|
hiringOrganization?: unknown;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
function findPostingNode(node: unknown): JobPostingNode | null {
|
|||
|
|
if (Array.isArray(node)) {
|
|||
|
|
for (const child of node) {
|
|||
|
|
const found = findPostingNode(child);
|
|||
|
|
if (found) return found;
|
|||
|
|
}
|
|||
|
|
return null;
|
|||
|
|
}
|
|||
|
|
if (!node || typeof node !== "object") return null;
|
|||
|
|
if ("@type" in node) {
|
|||
|
|
const t = node["@type"];
|
|||
|
|
const isPosting = t === "JobPosting" || (Array.isArray(t) && t.includes("JobPosting"));
|
|||
|
|
if (isPosting) {
|
|||
|
|
return {
|
|||
|
|
title: "title" in node ? node.title : undefined,
|
|||
|
|
description: "description" in node ? node.description : undefined,
|
|||
|
|
hiringOrganization: "hiringOrganization" in node ? node.hiringOrganization : undefined,
|
|||
|
|
};
|
|||
|
|
}
|
|||
|
|
}
|
|||
|
|
if ("@graph" in node) return findPostingNode(node["@graph"]);
|
|||
|
|
return null;
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
/** Strip an HTML document (or fragment) down to readable plain text. */
|
|||
|
|
export function stripHtml(html: string): string {
|
|||
|
|
let t = html
|
|||
|
|
.replace(/<!--[\s\S]*?-->/g, " ")
|
|||
|
|
.replace(/<(script|style|noscript|template|svg|head)\b[\s\S]*?<\/\1>/gi, " ")
|
|||
|
|
.replace(/<(br|\/p|\/div|\/li|\/tr|\/h[1-6]|\/section|\/article|\/ul|\/ol)[^>]*>/gi, "\n")
|
|||
|
|
.replace(/<li[^>]*>/gi, "\n- ")
|
|||
|
|
.replace(/<[^>]+>/g, " ");
|
|||
|
|
return tidy(decodeEntities(t));
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
const NAMED_ENTITIES: Record<string, string> = {
|
|||
|
|
amp: "&",
|
|||
|
|
lt: "<",
|
|||
|
|
gt: ">",
|
|||
|
|
quot: '"',
|
|||
|
|
apos: "'",
|
|||
|
|
nbsp: " ",
|
|||
|
|
ndash: "–",
|
|||
|
|
mdash: "—",
|
|||
|
|
bull: "•",
|
|||
|
|
hellip: "…",
|
|||
|
|
rsquo: "'",
|
|||
|
|
lsquo: "'",
|
|||
|
|
rdquo: '"',
|
|||
|
|
ldquo: '"',
|
|||
|
|
};
|
|||
|
|
|
|||
|
|
function decodeEntities(s: string): string {
|
|||
|
|
return s.replace(/&(#x?[0-9a-f]+|[a-z]+);/gi, (whole, code: string) => {
|
|||
|
|
if (code.startsWith("#x") || code.startsWith("#X"))
|
|||
|
|
return String.fromCodePoint(Number.parseInt(code.slice(2), 16));
|
|||
|
|
if (code.startsWith("#")) return String.fromCodePoint(Number.parseInt(code.slice(1), 10));
|
|||
|
|
return NAMED_ENTITIES[code.toLowerCase()] ?? whole;
|
|||
|
|
});
|
|||
|
|
}
|
|||
|
|
|
|||
|
|
/** Normalize whitespace: collapse runs, trim lines, drop blank-line stacks. */
|
|||
|
|
function tidy(s: string): string {
|
|||
|
|
return s
|
|||
|
|
.replace(/\r/g, "")
|
|||
|
|
.split("\n")
|
|||
|
|
.map((line) => line.replace(/[ \t\u00a0]+/g, " ").trim())
|
|||
|
|
.join("\n")
|
|||
|
|
.replace(/\n{3,}/g, "\n\n")
|
|||
|
|
.trim();
|
|||
|
|
}
|