46 lines
1.3 KiB
TypeScript
46 lines
1.3 KiB
TypeScript
import * as cheerio from "cheerio";
|
|
|
|
/**
|
|
* Fetch and extract the page title from a URL.
|
|
*/
|
|
export async function fetchUrlTitle(targetUrl: string): Promise<string | null> {
|
|
try {
|
|
const controller = new AbortController();
|
|
const timeout = setTimeout(() => controller.abort(), 2500);
|
|
|
|
const res = await fetch(targetUrl, {
|
|
signal: controller.signal,
|
|
headers: {
|
|
"User-Agent":
|
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
|
Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
|
},
|
|
});
|
|
|
|
clearTimeout(timeout);
|
|
|
|
if (!res.ok) return null;
|
|
|
|
const html = await res.text();
|
|
const $ = cheerio.load(html);
|
|
|
|
// 1. Try standard title tag
|
|
let title = $("title").first().text().trim();
|
|
|
|
// 2. Fallback to OpenGraph title
|
|
if (!title) {
|
|
title = $('meta[property="og:title"]').attr("content")?.trim() || "";
|
|
}
|
|
|
|
// 3. Fallback to Twitter title
|
|
if (!title) {
|
|
title = $('meta[name="twitter:title"]').attr("content")?.trim() || "";
|
|
}
|
|
|
|
return title ? title.slice(0, 255) : null;
|
|
} catch (error) {
|
|
console.warn(`Could not fetch title for ${targetUrl}:`, error);
|
|
return null;
|
|
}
|
|
}
|