diff --git a/backend/src/ingestion/adapters/base.ts b/backend/src/ingestion/adapters/base.ts index cb79d61..723dfd7 100644 --- a/backend/src/ingestion/adapters/base.ts +++ b/backend/src/ingestion/adapters/base.ts @@ -9,6 +9,8 @@ export interface FetchedItem { videos: { url: string; provider?: string; embedHtml?: string }[]; link: string; publishedAt: string; + /** Set by the Nitter adapter only — carries the tweet's author info through to ContentItem.tweet. */ + tweet?: { id: string; authorName: string; authorHandle: string; avatarUrl: string | null }; raw: unknown; } @@ -40,6 +42,7 @@ export function toContentItem(source: Source, item: FetchedItem): Omit>(); + +const USER_AGENT = 'Mozilla/5.0 (compatible; HomefeedBot/1.0; self-hosted RSS reader)'; +const FETCH_TIMEOUT_MS = 10_000; + +/** + * Shape based on the publicly documented FixTweet/fxtwitter API + * (https://github.com/FixTweet/FxTwitter) — NOT verified against a live response in + * this environment (outbound access to api.fxtwitter.com is blocked here). Every field + * below is read with optional chaining and a fallback in fetchFxTwitter's caller, so a + * shape mismatch degrades gracefully to RSS-only data rather than breaking ingestion. + * Verify against a real `curl https://api.fxtwitter.com/2/status/` response and + * adjust the field paths here if they don't match. + */ +interface FxTweet { + text?: string; + created_timestamp?: number; + author?: { name?: string; screen_name?: string; avatar_url?: string }; + media?: { photos?: { url?: string }[] }; +} + +async function fetchFxTwitter(tweetId: string): Promise { + try { + const res = await fetch(`https://api.fxtwitter.com/2/status/${tweetId}`, { + headers: { 'User-Agent': USER_AGENT }, + signal: AbortSignal.timeout(FETCH_TIMEOUT_MS) + }); + if (!res.ok) return null; + const json = (await res.json()) as { tweet?: FxTweet }; + return json?.tweet ?? null; + } catch (err) { + logger.warn('nitter', `fxtwitter enrichment failed for tweet ${tweetId}: ${(err as Error).message}`); + return null; + } +} + +function extractTweetId(item: Parser.Item): string | null { + const guid = item.guid; + if (guid && /^\d+$/.test(guid)) return guid; + const fromLink = item.link?.match(/status\/(\d+)/)?.[1]; + return fromLink ?? null; +} + +/** + * A Nitter list/user RSS description is the item's own tweet content (a

plus + * optional ) followed, for retweets/quote-tweets, by a

wrapping the + * quoted tweet's own text/image/permalink. Only the part before that blockquote is this + * item's own content — nested quote-tweet rendering isn't part of the approved design. + */ +function ownContentHtml(descriptionHtml: string): string { + const cut = descriptionHtml.search(/|
]+src="([^"]+)"/i)?.[1] ?? null; +} + +export const nitterAdapter: SourceAdapter = { + async fetch(source: Source): Promise { + if (!source.url) return []; + + const feed = await parser.parseURL(source.url); + const items: FetchedItem[] = []; + + for (const item of feed.items) { + if (!item.link || !item.guid) continue; + const tweetId = extractTweetId(item); + if (!tweetId) { + logger.warn('nitter', `Couldn't extract a tweet ID from "${item.link}" — skipping`); + continue; + } + + // dc:creator is reliably the author of this item's own tweet text (rss-parser + // maps it to item.creator) — for a retweet, that's the original tweet's author, + // not whichever list member's retweet surfaced it in this feed. + const handle = (item.creator ?? '').replace(/^@/, '') || 'unknown'; + const ownHtml = ownContentHtml(item.content ?? ''); + const rssImageUrl = extractImageUrl(ownHtml); + + const enrichment = await fetchFxTwitter(tweetId); + + const authorName = enrichment?.author?.name ?? handle; + const avatarUrl = enrichment?.author?.avatar_url ?? null; + const text = enrichment?.text ?? ownHtml; + const photoUrl = enrichment?.media?.photos?.[0]?.url ?? rssImageUrl; + const publishedAt = enrichment?.created_timestamp + ? new Date(enrichment.created_timestamp * 1000).toISOString() + : (item.isoDate ?? item.pubDate ?? new Date().toISOString()); + + items.push({ + title: item.title || text.slice(0, 100), + summary: text.slice(0, 500), + body: text, + images: photoUrl ? [{ url: photoUrl }] : [], + videos: [], + link: item.link, + publishedAt, + tweet: { id: tweetId, authorName, authorHandle: handle, avatarUrl }, + raw: { rss: item, fxtwitter: enrichment } + }); + } + + return items; + } +}; diff --git a/backend/src/ingestion/poller.ts b/backend/src/ingestion/poller.ts index 525d7ca..5a13c89 100644 --- a/backend/src/ingestion/poller.ts +++ b/backend/src/ingestion/poller.ts @@ -5,6 +5,7 @@ import { rssAdapter } from './adapters/rss.js'; import { telegramAdapter } from './adapters/telegram.js'; import { apiAdapter } from './adapters/api.js'; import { youtubeAdapter } from './adapters/youtube.js'; +import { nitterAdapter } from './adapters/nitter.js'; import { toContentItem, type SourceAdapter, type FetchedItem } from './adapters/base.js'; import { fetchFullArticle } from './articleFetcher.js'; import type { Source } from '../storage/db/types.js'; @@ -14,6 +15,7 @@ const adapters: Record = { telegram: telegramAdapter, api: apiAdapter, youtube: youtubeAdapter, + nitter: nitterAdapter, custom: apiAdapter }; diff --git a/backend/src/pipeline/publish.ts b/backend/src/pipeline/publish.ts index 0b95f44..ad0bc98 100644 --- a/backend/src/pipeline/publish.ts +++ b/backend/src/pipeline/publish.ts @@ -85,12 +85,16 @@ export async function publishDirect(item: ContentItem): Promise { const video = item.videos[0] ? { url: item.videos[0].url, provider: item.videos[0].provider, embedUrl: item.videos[0].embedHtml, sourceItemId: item.id } : null; + const tweet = item.tweet + ? { authorName: item.tweet.authorName, authorHandle: item.tweet.authorHandle, avatarUrl: item.tweet.avatarUrl, sourceItemId: item.id } + : null; const article = await articles.insertArticle({ title: item.title, body: item.body || item.summary, heroImage, video, + tweet, category, geo: item.geo, eventId: item.eventId, @@ -192,6 +196,7 @@ export async function publishCluster( body, heroImage, video, + tweet: null, // tweets never reach clustering — see priorityQueue.ts's direct-publish bypass category, geo, eventId: opts.eventId ?? items[0]?.eventId ?? null, diff --git a/backend/src/queue/priorityQueue.ts b/backend/src/queue/priorityQueue.ts index 181809e..12db2da 100644 --- a/backend/src/queue/priorityQueue.ts +++ b/backend/src/queue/priorityQueue.ts @@ -77,19 +77,22 @@ export async function runSynthesisCycle(provider: InferenceProvider, settings: G const items = contentItemsDb.unclusteredItemsExcludingSources(eventSourceIds); if (items.length === 0) return 0; - // YouTube videos never get LLM-merged with anything — each is always its own - // article (title/video/date/description), same shape whether the AI service is up - // or not. Route them straight to publishDirect, same as the no-AI passthrough path. - const youtubeSourceIds = new Set(sourcesDb.listSources().filter((s) => s.type === 'youtube').map((s) => s.id)); - const [youtubeItems, mergeableItems] = partition(items, (item) => youtubeSourceIds.has(item.sourceId)); + // YouTube videos and Nitter tweets never get LLM-merged with anything else — each + // is always its own article, same shape whether the AI service is up or not. Route + // them straight to publishDirect, same as the no-AI passthrough path. + const directPublishSourceIds = new Set( + sourcesDb.listSources().filter((s) => s.type === 'youtube' || s.type === 'nitter').map((s) => s.id) + ); + const [directItems, mergeableItems] = partition(items, (item) => directPublishSourceIds.has(item.sourceId)); let publishedDirect = 0; - for (const item of youtubeItems) { + for (const item of directItems) { try { const article = await publishDirect(item); contentItemsDb.assignCluster([item.id], article.id); publishedDirect++; - logger.info('synthesis', `Published "${article.title}" directly (YouTube)`); + const source = sourcesDb.getSource(item.sourceId); + logger.info('synthesis', `Published "${article.title}" directly (${source?.type ?? 'unknown'})`); } catch (err) { logger.error('synthesis', `Direct publish failed for "${item.title}": ${(err as Error).message}`); } diff --git a/backend/src/storage/db/articles.ts b/backend/src/storage/db/articles.ts index 1047542..17d5bbe 100644 --- a/backend/src/storage/db/articles.ts +++ b/backend/src/storage/db/articles.ts @@ -21,7 +21,8 @@ function rowToArticle(row: any): MergedArticle { threadId: row.thread_id, previousArticleId: row.previous_article_id, nextArticleId: row.next_article_id, - topStories: !!row.top_stories + topStories: !!row.top_stories, + tweet: row.tweet ? JSON.parse(row.tweet) : null }; } @@ -29,8 +30,8 @@ export function insertArticle(article: Omit): MergedArticle const id = `art-${randomUUID()}`; db.prepare( `INSERT INTO merged_articles - (id, title, body, hero_image, video, category, geo, event_id, source_count, sources, published_at, updated_at, merge_confidence, tags, thread_id, previous_article_id, next_article_id, top_stories) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + (id, title, body, hero_image, video, category, geo, event_id, source_count, sources, published_at, updated_at, merge_confidence, tags, thread_id, previous_article_id, next_article_id, top_stories, tweet) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` ).run( id, article.title, @@ -49,7 +50,8 @@ export function insertArticle(article: Omit): MergedArticle article.threadId, article.previousArticleId, article.nextArticleId, - article.topStories ? 1 : 0 + article.topStories ? 1 : 0, + article.tweet ? JSON.stringify(article.tweet) : null ); if (article.previousArticleId) { db.prepare('UPDATE merged_articles SET next_article_id = ? WHERE id = ?').run(id, article.previousArticleId); diff --git a/backend/src/storage/db/contentItems.ts b/backend/src/storage/db/contentItems.ts index e33a35b..f4942b2 100644 --- a/backend/src/storage/db/contentItems.ts +++ b/backend/src/storage/db/contentItems.ts @@ -20,6 +20,7 @@ function rowToItem(row: any): ContentItem { embedding: row.embedding ? JSON.parse(row.embedding) : null, eventId: row.event_id, clusterId: row.cluster_id, + tweet: row.tweet ? JSON.parse(row.tweet) : null, raw: row.raw ? JSON.parse(row.raw) : null }; } @@ -28,8 +29,8 @@ export function insertContentItem(item: Omit): ContentItem { const id = `ci-${randomUUID()}`; db.prepare( `INSERT INTO content_items - (id, source_id, type, title, summary, body, images, videos, link, published_at, fetched_at, tags, geo, embedding, event_id, cluster_id, raw) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` + (id, source_id, type, title, summary, body, images, videos, link, published_at, fetched_at, tags, geo, embedding, event_id, cluster_id, tweet, raw) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)` ).run( id, item.sourceId, @@ -47,6 +48,7 @@ export function insertContentItem(item: Omit): ContentItem { item.embedding ? JSON.stringify(item.embedding) : null, item.eventId, item.clusterId, + item.tweet ? JSON.stringify(item.tweet) : null, item.raw ? JSON.stringify(item.raw) : null ); return { ...item, id }; diff --git a/backend/src/storage/db/index.ts b/backend/src/storage/db/index.ts index d14223b..850769f 100644 --- a/backend/src/storage/db/index.ts +++ b/backend/src/storage/db/index.ts @@ -58,6 +58,7 @@ export function migrate() { embedding TEXT, -- JSON float array event_id TEXT, cluster_id TEXT, -- set once assigned to a cluster awaiting synthesis + tweet TEXT, -- JSON {id, authorName, authorHandle, avatarUrl}, nitter-sourced items only raw TEXT -- JSON, original payload ); CREATE INDEX IF NOT EXISTS idx_content_items_source ON content_items(source_id); @@ -82,7 +83,8 @@ export function migrate() { thread_id TEXT NOT NULL, previous_article_id TEXT, next_article_id TEXT, - top_stories INTEGER NOT NULL DEFAULT 0 -- true if any contributing source opted into "Push to Top Stories?" + top_stories INTEGER NOT NULL DEFAULT 0, -- true if any contributing source opted into "Push to Top Stories?" + tweet TEXT -- JSON {authorName, authorHandle, avatarUrl, sourceItemId}, nitter-sourced articles only ); CREATE INDEX IF NOT EXISTS idx_articles_published ON merged_articles(published_at); CREATE INDEX IF NOT EXISTS idx_articles_thread ON merged_articles(thread_id); @@ -187,6 +189,12 @@ export function migrate() { if (!hasColumn('merged_articles', 'top_stories')) { db.exec('ALTER TABLE merged_articles ADD COLUMN top_stories INTEGER NOT NULL DEFAULT 0'); } + if (!hasColumn('content_items', 'tweet')) { + db.exec('ALTER TABLE content_items ADD COLUMN tweet TEXT'); + } + if (!hasColumn('merged_articles', 'tweet')) { + db.exec('ALTER TABLE merged_articles ADD COLUMN tweet TEXT'); + } // Seed default categories if none exist yet. "News" sits right under "Top stories" — // general news sources belong here, not on "Top stories" itself, which isn't a real diff --git a/backend/src/storage/db/types.ts b/backend/src/storage/db/types.ts index d62eb3c..778ff42 100644 --- a/backend/src/storage/db/types.ts +++ b/backend/src/storage/db/types.ts @@ -1,7 +1,7 @@ export interface Source { id: string; name: string; - type: 'rss' | 'api' | 'telegram' | 'youtube' | 'custom'; + type: 'rss' | 'api' | 'telegram' | 'youtube' | 'nitter' | 'custom'; category: string[]; url: string | null; config: Record; @@ -31,6 +31,8 @@ export interface ContentItem { embedding: number[] | null; eventId: string | null; clusterId: string | null; + /** Nitter-sourced items only — null for everything else. */ + tweet: { id: string; authorName: string; authorHandle: string; avatarUrl: string | null } | null; raw: unknown; } @@ -47,6 +49,8 @@ export interface MergedArticle { body: string; heroImage: { url: string; sourceItemId: string; selectionReason: string } | null; video: { url: string; provider?: string; embedUrl?: string; sourceItemId: string } | null; + /** Nitter-sourced articles only — the embed card's author info (see TweetCard.svelte). Never set alongside video. */ + tweet: { authorName: string; authorHandle: string; avatarUrl: string | null; sourceItemId: string } | null; category: string[]; geo: string | null; eventId: string | null; diff --git a/frontend/src/lib/adminTypes.ts b/frontend/src/lib/adminTypes.ts index 9d0b128..5b7ba63 100644 --- a/frontend/src/lib/adminTypes.ts +++ b/frontend/src/lib/adminTypes.ts @@ -32,7 +32,7 @@ export interface AdminSettings { export interface AdminSource { id: string; name: string; - type: 'rss' | 'api' | 'telegram' | 'youtube' | 'custom'; + type: 'rss' | 'api' | 'telegram' | 'youtube' | 'nitter' | 'custom'; category: string[]; url: string; config?: Record; diff --git a/frontend/src/lib/components/ArticleListRow.svelte b/frontend/src/lib/components/ArticleListRow.svelte index 1f37e95..d32248b 100644 --- a/frontend/src/lib/components/ArticleListRow.svelte +++ b/frontend/src/lib/components/ArticleListRow.svelte @@ -2,6 +2,7 @@ import type { MergedArticle } from '$lib/types'; import { timeAgo, exactTime, excerpt } from '$lib/format'; import { resolveMediaUrl } from '$lib/config'; + import TweetCard from './TweetCard.svelte'; let { article }: { article: MergedArticle } = $props(); @@ -12,28 +13,32 @@ ); - - {#if article.heroImage} - - {:else} -
- {/if} +{#if article.tweet} + +{:else} +
+ {#if article.heroImage} + + {:else} +
+ {/if} -
-
- {article.category[0] ?? ''} - · - {sourceLabel} - {#if article.video} +
+
+ {article.category[0] ?? ''} · - ▶ Video - {/if} + {sourceLabel} + {#if article.video} + · + ▶ Video + {/if} +
+
{article.title}
+
{excerpt(article.body)}
+
{timeAgo(article.publishedAt)} · {exactTime(article.publishedAt)}
-
{article.title}
-
{excerpt(article.body)}
-
{timeAgo(article.publishedAt)} · {exactTime(article.publishedAt)}
-
-
+ +{/if} diff --git a/frontend/src/lib/components/admin/SourcesTab.svelte b/frontend/src/lib/components/admin/SourcesTab.svelte index f3f3b11..bb83676 100644 --- a/frontend/src/lib/components/admin/SourcesTab.svelte +++ b/frontend/src/lib/components/admin/SourcesTab.svelte @@ -144,7 +144,17 @@ } const typeIcon = (type: string) => - type === 'rss' ? '⟳' : type === 'telegram' ? '✈' : type === 'youtube' ? '▶' : type === 'api' ? '⇄' : '•'; + type === 'rss' + ? '⟳' + : type === 'telegram' + ? '✈' + : type === 'youtube' + ? '▶' + : type === 'nitter' + ? '🐦' + : type === 'api' + ? '⇄' + : '•';
@@ -163,10 +173,13 @@ + {#if form.type === 'youtube'} + {:else if form.type === 'nitter'} + {:else} {/if} diff --git a/frontend/src/lib/types.ts b/frontend/src/lib/types.ts index 843cd8a..625b4f7 100644 --- a/frontend/src/lib/types.ts +++ b/frontend/src/lib/types.ts @@ -14,6 +14,7 @@ export interface MergedArticle { body: string; heroImage: { url: string; sourceItemId: string; selectionReason: string } | null; video: { url: string; provider?: string; embedUrl?: string; sourceItemId: string } | null; + tweet: { authorName: string; authorHandle: string; avatarUrl: string | null; sourceItemId: string } | null; category: string[]; geo: string | null; eventId: string | null; diff --git a/frontend/src/routes/article/[id]/+page.svelte b/frontend/src/routes/article/[id]/+page.svelte index 3788cd0..418259d 100644 --- a/frontend/src/routes/article/[id]/+page.svelte +++ b/frontend/src/routes/article/[id]/+page.svelte @@ -2,6 +2,7 @@ import type { PageData } from './$types'; import { timeAgo, exactTime } from '$lib/format'; import { resolveMediaUrl } from '$lib/config'; + import TweetCard from '$lib/components/TweetCard.svelte'; let { data }: { data: PageData } = $props(); const a = $derived(data.article); @@ -22,9 +23,22 @@ {a.category.join(', ')}
-

{a.title}

+ {#if a.tweet} + +
+ Published {timeAgo(a.publishedAt)} · {exactTime(a.publishedAt)} +
+ + + + {#if a.sources[0]} + View original tweet → + {/if} + {:else if a.video?.provider === 'youtube'} +

{a.title}

- {#if a.video?.provider === 'youtube'} @@ -46,6 +60,8 @@

{paragraph}

{/each} {:else} +

{a.title}

+
Published {timeAgo(a.publishedAt)} · {exactTime(a.publishedAt)} {#if a.updatedAt !== a.publishedAt} @@ -180,6 +196,12 @@ height: 100%; border: none; } + .view-original { + display: inline-block; + font-size: 13px; + color: var(--text-accent); + margin-bottom: 20px; + } .tags { display: flex; gap: 8px;