fix(jobindex-search): decode hex HTML entities in CLI output (#56)

decodeHtmlEntities (duplicated in src/helpers.ts and
src/commands/detail.ts) only handled decimal numeric character
references (é); the equally valid hexadecimal form (é) fell
through undecoded and surfaced as raw text in titles, companies,
locations and descriptions. This bites Danish content especially
(ae/o/aa often arrive as entities). It also used String.fromCharCode,
which corrupts supplementary-plane code points (e.g. emoji, U+1F600).

Add a hexadecimal numeric-entity rule and route both decimal and hex
through a fromCodePoint-based helper with a valid-range guard, in both
copies. Add network-free unit tests via the exported parseJobCards.
This commit is contained in:
Yiğit ERDOĞAN
2026-07-07 19:41:23 +02:00
committed by GitHub
parent b27a3b5e81
commit 844b245583
3 changed files with 70 additions and 2 deletions
@@ -19,6 +19,15 @@ interface DetailResult {
description: string | null
}
/**
* Convert a Unicode code point to a string. Uses `fromCodePoint` (not
* `fromCharCode`) so supplementary-plane code points (e.g. emoji, U+1F600)
* decode correctly, and drops out-of-range values instead of throwing.
*/
function numericEntity(cp: number): string {
return cp >= 0 && cp <= 0x10ffff ? String.fromCodePoint(cp) : ""
}
/**
* Decode HTML entities in text
*/
@@ -30,7 +39,9 @@ function decodeHtmlEntities(text: string): string {
.replace(/&quot;/g, '"')
.replace(/&#39;/g, "'")
.replace(/&apos;/g, "'")
.replace(/&#(\d+);/g, (_, code) => String.fromCharCode(parseInt(code, 10)))
// Numeric character references: decimal (&#233;) and hexadecimal (&#xE9;).
.replace(/&#(\d+);/g, (_, dec) => numericEntity(parseInt(dec, 10)))
.replace(/&#[xX]([0-9a-fA-F]+);/g, (_, hex) => numericEntity(parseInt(hex, 16)))
.replace(/&nbsp;/g, " ")
}
@@ -180,6 +180,15 @@ export function parseSearchPage(html: string): SearchPageResult {
return { total, results }
}
/**
* Convert a Unicode code point to a string. Uses `fromCodePoint` (not
* `fromCharCode`) so supplementary-plane code points (e.g. emoji, U+1F600)
* decode correctly, and drops out-of-range values instead of throwing.
*/
function numericEntity(cp: number): string {
return cp >= 0 && cp <= 0x10ffff ? String.fromCodePoint(cp) : ""
}
/**
* Decode HTML entities in text
*/
@@ -191,7 +200,9 @@ function decodeHtmlEntities(text: string): string {
.replace(/&quot;/g, '"')
.replace(/&#39;/g, "'")
.replace(/&apos;/g, "'")
.replace(/&#(\d+);/g, (_, code) => String.fromCharCode(parseInt(code, 10)))
// Numeric character references: decimal (&#233;) and hexadecimal (&#xE9;).
.replace(/&#(\d+);/g, (_, dec) => numericEntity(parseInt(dec, 10)))
.replace(/&#[xX]([0-9a-fA-F]+);/g, (_, hex) => numericEntity(parseInt(hex, 16)))
.replace(/&nbsp;/g, " ")
}