← Files DefuddleARCHIVED FILE
skills/defuddle-web-content/references/source/src/extractors/hackernews.ts
9.36 KB · Oct 4, 2026 · 12:35 UTC
import { BaseExtractor, ExtractorOptions } from './_base';
import { ExtractorResult } from '../types/extractors';
import { serializeHTML, escapeHtml } from '../utils/dom';
import { buildComment, buildCommentTree, buildContentHtml, CommentData } from '../utils/comments';
interface StoryData {
id: string;
title: string;
url: string;
site: string;
score: string;
author: string;
date: string;
comments: string;
commentsUrl: string;
}
export class HackerNewsExtractor extends BaseExtractor {
private mainPost: Element | null;
private isCommentPage: boolean;
private isListingPage: boolean;
private mainComment: Element | null;
constructor(document: Document, url: string, schemaOrgData?: any, options?: ExtractorOptions) {
super(document, url, schemaOrgData, options);
this.mainPost = document.querySelector('.fatitem');
this.isListingPage = this.detectListingPage();
this.isCommentPage = this.detectCommentPage();
this.mainComment = this.isCommentPage ? this.findMainComment() : null;
}
private detectListingPage(): boolean {
if (this.mainPost) return false;
const stories = this.document.querySelectorAll('tr.athing');
return stories.length > 1;
}
private detectCommentPage(): boolean {
// Comment pages have an "on: <story title>" link but no story title row
return !!this.mainPost?.querySelector('.onstory') && !this.mainPost?.querySelector('.titleline');
}
private findMainComment(): Element | null {
// Use the tr.athing row which contains both the comment metadata (.comhead)
// and the comment text (.commtext). The .comment div alone doesn't include
// the author (.hnuser) or timestamp (.age) which are in the sibling .comhead.
return this.mainPost?.querySelector('tr.athing') || null;
}
canExtract(): boolean {
return !!this.mainPost || this.isListingPage;
}
extract(): ExtractorResult {
if (this.isListingPage) {
return this.extractListing();
}
const postContent = this.getPostContent();
const comments = this.options.includeReplies !== false ? this.extractComments() : '';
const contentHtml = this.createContentHtml(postContent, comments);
const postTitle = this.getPostTitle();
const postAuthor = this.getPostAuthor();
const description = this.createDescription();
const published = this.getPostDate();
return {
content: contentHtml,
contentHtml: contentHtml,
extractedContent: {
postId: this.getPostId(),
postAuthor,
},
variables: {
title: postTitle,
author: postAuthor,
site: 'Hacker News',
description,
published,
}
};
}
private getMoreLink(): { url: string; text: string } | null {
const moreLink = this.document.querySelector('.morelink');
if (!moreLink) return null;
const href = moreLink.getAttribute('href') || '';
return { url: href, text: moreLink.textContent?.trim() || 'More' };
}
private extractListing(): ExtractorResult {
const stories = this.extractStories();
const moreLink = this.getMoreLink();
const contentHtml = this.buildListingHtml(stories, moreLink);
const title = this.document.title?.replace(/\s*\|\s*Hacker News$/, '').trim() || 'Hacker News';
return {
content: contentHtml,
contentHtml: contentHtml,
extractedContent: {},
variables: {
title,
site: 'Hacker News',
}
};
}
private extractStories(): StoryData[] {
const storyRows = Array.from(this.document.querySelectorAll('tr.athing'));
const stories: StoryData[] = [];
for (const row of storyRows) {
const id = row.getAttribute('id') || '';
const titleEl = row.querySelector('.titleline a');
if (!titleEl) continue;
const title = titleEl.textContent?.trim() || '';
const storyUrl = titleEl.getAttribute('href') || '';
const site = row.querySelector('.sitestr')?.textContent?.trim() || '';
// The subtext row is the next sibling tr
const subRow = row.nextElementSibling;
const score = subRow?.querySelector('.score')?.textContent?.trim() || '';
const author = subRow?.querySelector('.hnuser')?.textContent?.trim() || '';
const ageEl = subRow?.querySelector('.age');
const timestamp = ageEl?.getAttribute('title') || '';
const date = timestamp.split('T')[0] || '';
const subLinks = subRow ? Array.from(subRow.querySelectorAll('td.subtext a')) : [];
const lastLink = subLinks[subLinks.length - 1];
const commentsText = lastLink?.textContent?.replace(/\u00a0/g, ' ').trim() || '';
const comments = /\d+\s*comment/.test(commentsText) ? commentsText : '';
const commentsUrl = id ? `https://news.ycombinator.com/item?id=${id}` : '';
stories.push({ id, title, url: storyUrl, site, score, author, date, comments, commentsUrl });
}
return stories;
}
private buildListingHtml(stories: StoryData[], moreLink: { url: string; text: string } | null): string {
if (stories.length === 0) return '';
const items = stories.map(story => {
let html = '<li>';
html += `<a href="${escapeHtml(story.url)}">${escapeHtml(story.title)}</a>`;
if (story.site) {
html += ` <small>(${escapeHtml(story.site)})</small>`;
}
const meta: string[] = [];
if (story.score) meta.push(escapeHtml(story.score));
if (story.author) meta.push(`by ${escapeHtml(story.author)}`);
if (story.comments) {
meta.push(`<a href="${escapeHtml(story.commentsUrl)}">${escapeHtml(story.comments)}</a>`);
}
if (meta.length > 0) {
html += `<br><small>${meta.join(' · ')}</small>`;
}
html += '</li>';
return html;
});
let html = `<ol>${items.join('')}</ol>`;
if (moreLink) {
html += `<p><a href="${escapeHtml(moreLink.url)}">${escapeHtml(moreLink.text)}</a></p>`;
}
return html;
}
private createContentHtml(postContent: string, comments: string): string {
return buildContentHtml('hackernews', postContent, comments);
}
private getPostContent(): string {
if (!this.mainPost) return '';
// If this is a comment page, use the comment as the main content
if (this.isCommentPage && this.mainComment) {
const author = this.mainComment.querySelector('.hnuser')?.textContent || '[deleted]';
const commtext = this.mainComment.querySelector('.commtext');
const commentText = commtext ? serializeHTML(commtext) : '';
const timeElement = this.mainComment.querySelector('.age');
const timestamp = timeElement?.getAttribute('title') || '';
const date = timestamp.split('T')[0] || '';
const points = this.mainComment.querySelector('.score')?.textContent?.trim() || '';
return buildComment({
author,
date,
content: commentText,
score: points || undefined,
});
}
// Otherwise handle regular post content
const titleRow = this.mainPost.querySelector('tr.athing');
const subRow = titleRow?.nextElementSibling;
const url = titleRow?.querySelector('.titleline a')?.getAttribute('href') || '';
let content = '';
if (url) {
content += `<p><a href="${url}" target="_blank">${url}</a></p>`;
}
const text = this.mainPost.querySelector('.toptext');
if (text) {
content += `<div class="post-text">${serializeHTML(text)}</div>`;
}
return content;
}
private extractComments(): string {
const comments = Array.from(this.document.querySelectorAll('tr.comtr'));
return this.processComments(comments);
}
private processComments(comments: Element[]): string {
const commentData: CommentData[] = [];
const processedIds = new Set<string>();
for (const comment of comments) {
const id = comment.getAttribute('id');
if (!id || processedIds.has(id)) continue;
processedIds.add(id);
const indent = comment.querySelector('.ind img')?.getAttribute('width') || '0';
const depth = parseInt(indent) / 40;
const commentText = comment.querySelector('.commtext');
const author = comment.querySelector('.hnuser')?.textContent || '[deleted]';
const timeElement = comment.querySelector('.age');
const points = comment.querySelector('.score')?.textContent?.trim() || '';
if (!commentText) continue;
const commentUrl = `https://news.ycombinator.com/item?id=${id}`;
const timestamp = timeElement?.getAttribute('title') || '';
const date = timestamp.split('T')[0] || '';
commentData.push({
author,
date,
content: serializeHTML(commentText),
depth,
score: points || undefined,
url: commentUrl,
});
}
return buildCommentTree(commentData);
}
private getPostId(): string {
const match = this.url.match(/id=(\d+)/);
return match?.[1] || '';
}
private getPostTitle(): string {
if (this.isCommentPage && this.mainComment) {
const author = this.mainComment.querySelector('.hnuser')?.textContent || '[deleted]';
const commentText = this.mainComment.querySelector('.commtext')?.textContent || '';
// Use first 50 characters of comment as title
const preview = commentText.trim().slice(0, 50) + (commentText.length > 50 ? '...' : '');
return `Comment by ${author}: ${preview}`;
}
return this.mainPost?.querySelector('.titleline')?.textContent?.trim() || '';
}
private getPostAuthor(): string {
return this.mainPost?.querySelector('.hnuser')?.textContent?.trim() || '';
}
private createDescription(): string {
const title = this.getPostTitle();
const author = this.getPostAuthor();
if (this.isCommentPage) {
return `Comment by ${author} on Hacker News`;
}
return `${title} - by ${author} on Hacker News`;
}
private getPostDate(): string {
if (!this.mainPost) return '';
const timeElement = this.mainPost.querySelector('.age');
const timestamp = timeElement?.getAttribute('title') || '';
return timestamp.split('T')[0] || '';
}
} SHA-256: cad27c2ab798f2c24ba307154feeb0d560000a33e6b1f01ef6c25c536f2828ef