// HTML / RSS text-extraction helpers for find-substack-post.js.
// Pure string functions (no network, no fs) — kept separate so the crawler
// stays focused on fetch/search flow. CommonJS so plain `node` works.
// Decode the HTML entities that appear in Substack titles/captions, including
// numeric (') and hex (—) forms. & is decoded last-ish but
// before nothing re-encodes it; ordering here is safe for our inputs.
function decodeEntities(s) {
return s
.replace(/&/g, "&")
.replace(/</g, "<")
.replace(/>/g, ">")
.replace(/"/g, '"')
.replace(/?39;|'/g, "'")
.replace(/ /g, " ")
.replace(/([0-9a-fA-F]+);/g, (_, n) => String.fromCodePoint(parseInt(n, 16)))
.replace(/(\d+);/g, (_, n) => String.fromCodePoint(+n))
.trim();
}
function stripTags(html) {
return decodeEntities(html.replace(/<[^>]+>/g, "")).replace(/\s+/g, " ").trim();
}
// Pull the first
(CDATA or plain) and from an RSS chunk.
function itemTitle(item) {
const m = item.match(/(?:)?<\/title>/);
return m ? decodeEntities(m[1].trim()) : "";
}
function itemLink(item) {
const m = item.match(/(?:)?<\/link>/);
return m ? m[1].trim() : "";
}
// Extract candidate topic titles from a post's TOC bullet list. ByteByteGo does
// not attach captions to images — the topic titles live only in the "in this
// issue" bullets. We can't reliably map image→title automatically (sponsor /
// video items interleave and break positional order), so we surface these
// candidates for the user to pick from. Light filtering keeps the list short:
// dedupe, drop sub-point explanations ("Term: long sentence") and over-long
// lines. The correct title is always present; the user selects it.
function extractCandidates(html) {
const seen = new Set();
const out = [];
const re = /
]*>\s*(?:
]*>)?([\s\S]*?)(?:<\/p>)?\s*<\/li>/g;
let m;
while ((m = re.exec(html))) {
const text = stripTags(m[1]);
if (!text) continue;
if (text.length < 6 || text.length > 70) continue; // titles are short; long lines are sub-point explanations
const key = text.toLowerCase();
if (seen.has(key)) continue; // content is duplicated in the page
seen.add(key);
out.push(text);
}
return out;
}
// If the UUID sits inside a …, return that figure's
// text. Cover images live in (no figure) → "".
function captionForUuid(item, id) {
const at = item.indexOf(id);
if (at === -1) return "";
const figStart = item.lastIndexOf("", at);
if (figEnd === -1) return "";
const figure = item.slice(figStart, figEnd);
const cap = figure.match(/]*>([\s\S]*?)<\/figcaption>/);
return cap ? stripTags(cap[1]) : "";
}
// Extract a post title from server-rendered post HTML (og:title preferred).
function postTitleFromHtml(html) {
const og = html.match(/]+property=["']og:title["'][^>]+content=["']([^"']+)["']/i);
if (og) return decodeEntities(og[1]);
const h1 = html.match(/
]*>([\s\S]*?)<\/h1>/i);
if (h1) return stripTags(h1[1]);
const t = html.match(/([\s\S]*?)<\/title>/i);
return t ? decodeEntities(t[1].trim()) : "";
}
module.exports = {
decodeEntities,
stripTags,
itemTitle,
itemLink,
extractCandidates,
captionForUuid,
postTitleFromHtml,
};