]*class="[^"]*${escaped}[^"]*"[^>]*>`, 'i')
+ const open = openRe.exec(html)
+ if (!open) return null
+
+ let i = open.index + open[0].length
+ let depth = 1
+
+ while (depth > 0 && i < html.length) {
+ const nextOpen = html.indexOf('
', i)
+
+ if (nextClose === -1) return null
+
+ if (nextOpen !== -1 && nextOpen < nextClose) {
+ depth++
+ i = nextOpen + 4
+ } else {
+ depth--
+ i = nextClose + 6
+ }
+ }
+
+ return html.slice(open.index + open[0].length, i - 6)
+}
+
/**
* Convert a Unicode code point to a string. Uses `fromCodePoint` (not
* `fromCharCode`) so supplementary-plane code points (e.g. emoji, U+1F600)
@@ -180,11 +211,11 @@ export function parseJobDetail(html: string, id: string): JobDetail {
// Rich description block. Keep paragraph/line breaks as newlines.
let description: string | null = null
- const desc = html.match(
- /class="(?:show-more-less-html__markup|description__text[^"]*)"[^>]*>([\s\S]*?)<\/div>/i,
- )
- if (desc) {
- const withBreaks = desc[1]
+ const descHtml =
+ extractDivContent(html, "show-more-less-html__markup") ??
+ extractDivContent(html, "description__text")
+ if (descHtml) {
+ const withBreaks = descHtml
.replace(/<\s*br\s*\/?>/gi, "\n")
.replace(/<\/(p|li|ul|ol|div|h\d)>/gi, "\n")
description = decodeHtmlEntities(stripTags(withBreaks)).replace(/\n{3,}/g, "\n\n").trim() || null
diff --git a/.agents/skills/linkedin-search/cli/tests/parsing.test.ts b/.agents/skills/linkedin-search/cli/tests/parsing.test.ts
index 49d3115..792659c 100644
--- a/.agents/skills/linkedin-search/cli/tests/parsing.test.ts
+++ b/.agents/skills/linkedin-search/cli/tests/parsing.test.ts
@@ -1,5 +1,5 @@
import { describe, test, expect } from "bun:test";
-import { parseJobCards, parseJobDetail } from "../src/helpers";
+import { parseJobCards, parseJobDetail, extractDivContent } from "../src/helpers";
// Minimal search-card markup: parseJobCards splits on the job-posting URN and
// needs an id, a base-search-card__title, and a full-link. Everything else is
@@ -53,3 +53,61 @@ describe("decodeHtmlEntities (via parseJobDetail)", () => {
expect(job.title).toBe("Señor Engineer");
});
});
+
+describe("extractDivContent", () => {
+ test("extracts content from simple div", () => {
+ const html = '
Simple text
';
+ expect(extractDivContent(html, "description__text")).toBe("Simple text");
+ });
+
+ test("extracts content with nested divs — the regression case", () => {
+ const html = `
+
Requirements:
+
+
About Us:
+
We are...
+
`;
+ expect(extractDivContent(html, "description__text")).toBe(
+ '\n
Requirements:
\n
\n
About Us:
\n
We are...
\n ',
+ );
+ });
+
+ test("returns null when class not found", () => {
+ expect(extractDivContent("
no class
", "nonexistent")).toBeNull();
+ });
+
+ test("works with show-more-less-html__markup class", () => {
+ const html = '
LinkedIn content
';
+ expect(extractDivContent(html, "show-more-less-html__markup")).toBe("LinkedIn content");
+ });
+
+ test("handles deeply nested divs (3 levels)", () => {
+ const html = `
`;
+ expect(extractDivContent(html, "description__text")).toBe(
+ '\n
\n ',
+ );
+ });
+
+ test("handles empty content", () => {
+ const html = '
';
+ expect(extractDivContent(html, "description__text")).toBe("");
+ });
+
+ test("parseJobDetail uses extractDivContent and preserves full description", () => {
+ const html = `
+
Requirements:
+
+
About Us:
+
We are hiring!
+
`;
+ const job = parseJobDetail(html, "999");
+ expect(job.description).toContain("Requirements:");
+ expect(job.description).toContain("5 years Python");
+ expect(job.description).toContain("About Us:");
+ expect(job.description).toContain("We are hiring!");
+ });
+});