← Files DefuddleARCHIVED FILE

skills/defuddle-web-content/references/source/src/extractors/_base.ts

1.42 KB · Oct 5, 2026 · 18:35 UTC

↓ Download file

import { ExtractorResult, ExtractorVariables, ExtractedContent } from '../types/extractors';

export interface ExtractorOptions {
	includeReplies?: boolean | 'extractors';
	language?: string;
	fetch?: typeof globalThis.fetch;
}

export abstract class BaseExtractor {
	protected document: Document;
	protected url: string;
	protected schemaOrgData?: any;
	protected options: ExtractorOptions;

	constructor(document: Document, url: string, schemaOrgData?: any, options?: ExtractorOptions) {
		this.document = document;
		this.url = url;
		this.schemaOrgData = schemaOrgData;
		this.options = options || {};
	}

	protected get fetch(): typeof globalThis.fetch {
		const fn = this.options.fetch || globalThis.fetch;
		return fn.bind(globalThis);
	}

	abstract canExtract(): boolean;
	abstract extract(): ExtractorResult;

	/**
	 * Generate a title from the post description text, falling back to
	 * "Post by {author}" if the description is empty.
	 */
	protected postTitle(author: string, site: string): string {
		return `Post by ${author} on ${site}`;
	}

	canExtractAsync(): boolean {
		return false;
	}

	/**
	 * When true, parseAsync() will prefer extractAsync() over extract(),
	 * even if sync extraction produces content. Use this when the async
	 * path provides strictly better results (e.g. YouTube transcripts).
	 */
	prefersAsync(): boolean {
		return false;
	}

	async extractAsync(): Promise<ExtractorResult> {
		return this.extract();
	}
} 

SHA-256: f68941ee6d7176e6afe2d117b524d909510412eb7378e14b89c40a580e59561a