mirror of
https://github.com/MadsLorentzen/ai-job-search.git
synced 2026-09-17 08:36:25 +00:00
fix(portals): depth-track div extraction so nested job descriptions aren't truncated (#204)
The jobindex and linkedin detail parsers matched description containers with a non-greedy regex that stops at the first inner </div>, so any posting whose description contains nested divs was silently truncated (jobindex dropped later sections; linkedin dropped everything after the first block). Replaces the regex with a depth-tracked extractDivContent scanner that walks div open/close markers to the matching close. Verified: truncation bug reproduced against real markup fixtures, depth arithmetic correct (no off-by-one/infinite-loop), 28 tests pass network-free, no regression on non-nested divs. Malformed-HTML over-grabs rather than truncates - the safer failure, cleaned by downstream stripTags/decode. By @oscarbol09.
This commit is contained in:
@@ -1,5 +1,5 @@
|
||||
import { describe, test, expect } from "bun:test";
|
||||
import { parseJobCards } from "../src/helpers";
|
||||
import { parseJobCards, extractDivContent } from "../src/helpers";
|
||||
|
||||
// Minimal jobad-wrapper markup: parseJobCards splits on `jobad-wrapper-<id>`
|
||||
// and reads the title from the <h4><a href>…</a></h4> and the company from the
|
||||
@@ -44,3 +44,55 @@ describe("decodeHtmlEntities (via parseJobCards)", () => {
|
||||
expect(c.company).toBe("Nørrebro ApS");
|
||||
});
|
||||
});
|
||||
|
||||
describe("extractDivContent", () => {
|
||||
test("extracts content from simple div", () => {
|
||||
const html = '<div class="job-text">Simple text</div>';
|
||||
expect(extractDivContent(html, "job-text")).toBe("Simple text");
|
||||
});
|
||||
|
||||
test("extracts content with nested divs — the regression case", () => {
|
||||
const html = `<div class="job-text">
|
||||
<div class="highlight">First section</div>
|
||||
<div class="details">Second section with important info</div>
|
||||
</div>`;
|
||||
expect(extractDivContent(html, "job-text")).toBe(
|
||||
'\n <div class="highlight">First section</div>\n <div class="details">Second section with important info</div>\n ',
|
||||
);
|
||||
});
|
||||
|
||||
test("returns null when class not found", () => {
|
||||
expect(extractDivContent("<div>no class</div>", "nonexistent")).toBeNull();
|
||||
});
|
||||
|
||||
test("works with extra attributes on the div", () => {
|
||||
const html = '<div id="desc" class="job-text" data-x="1">Content</div>';
|
||||
expect(extractDivContent(html, "job-text")).toBe("Content");
|
||||
});
|
||||
|
||||
test("handles deeply nested divs (3 levels)", () => {
|
||||
const html = `<div class="job-text">
|
||||
<div>
|
||||
<div>Deep content</div>
|
||||
</div>
|
||||
</div>`;
|
||||
expect(extractDivContent(html, "job-text")).toBe(
|
||||
'\n <div>\n <div>Deep content</div>\n </div>\n ',
|
||||
);
|
||||
});
|
||||
|
||||
test("handles empty content", () => {
|
||||
const html = '<div class="job-text"></div>';
|
||||
expect(extractDivContent(html, "job-text")).toBe("");
|
||||
});
|
||||
|
||||
test("handles br and other non-div tags", () => {
|
||||
const html = '<div class="job-text">Line1<br>Line2<br>Line3</div>';
|
||||
expect(extractDivContent(html, "job-text")).toBe("Line1<br>Line2<br>Line3");
|
||||
});
|
||||
|
||||
test("escapes special regex characters in class name", () => {
|
||||
const html = '<div class="job-text (special)">Content</div>';
|
||||
expect(extractDivContent(html, "job-text (special)")).toBe("Content");
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user