From ecfcce9eb17e1a92dc15a6dc89e8a110d8a1273e Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 27 Jul 2026 14:29:45 +0000 Subject: [PATCH] Have the model synthesize a real title instead of truncating the body MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every article title ended in "…" because there was never an actual title — deriveTitle() just took the body's first paragraph and cut it at 97 characters. The AI was never asked for a headline at all. Both system prompts now ask for a response in three parts (headline, then the article/recap, then tags), each separated by a delimiter. parseResult() extracts all three; if the model doesn't follow the format at all, it falls back to the old truncated-first-line heuristic rather than breaking. Delimiter matching is now a loose regex instead of an exact string — production had already shown a small model reproducing "---TAGS---" inexactly (e.g. "---\n\nTAGS---"), which the old exact-string split missed entirely and leaked into the published body. Same tolerance now applies to the new title delimiter. publishCluster uses the synthesized title directly; publishEventRecap uses it too, falling back to the previous ": recap" format only if the model returns an empty title. Verified: exact-format output, sloppy-delimiter output, and no-delimiter-at-all output all parse into sensible {title, body, tags}; a full runSynthesisCycle pass against a mock provider publishes an article with the real synthesized headline as its title. --- backend/src/pipeline/publish.ts | 14 +++----- backend/src/pipeline/synthesis.ts | 59 +++++++++++++++++++++++-------- 2 files changed, 49 insertions(+), 24 deletions(-) diff --git a/backend/src/pipeline/publish.ts b/backend/src/pipeline/publish.ts index b5348ef..0c18fcf 100644 --- a/backend/src/pipeline/publish.ts +++ b/backend/src/pipeline/publish.ts @@ -36,12 +36,6 @@ function anyPushesToTopStories(items: ContentItem[]): boolean { return items.some((item) => sources.getSource(item.sourceId)?.pushToTopStories ?? false); } -/** Takes the first line of the synthesized body as a working title until a dedicated title-generation step exists. */ -function deriveTitle(body: string): string { - const firstLine = body.split('\n')[0]; - return firstLine.length > 100 ? firstLine.slice(0, 97) + '…' : firstLine; -} - /** * Resolves the hero image for a regular (non-tweet) article: try the best candidate * from the source items, download and locally host it; if there isn't one, fall back @@ -362,7 +356,7 @@ export async function publishCluster( const items = cluster.items; const sourceNames = new Map(items.map((item) => [item.sourceId, sources.getSource(item.sourceId)?.name ?? 'Unknown source'])); - const { body, tagLabels } = await synthesizeArticle(provider, settings.selectedModels.synthesis, items, sourceNames, settings); + const { title, body, tagLabels } = await synthesizeArticle(provider, settings.selectedModels.synthesis, items, sourceNames, settings); const resolvedTags = []; for (const label of tagLabels) { @@ -418,7 +412,7 @@ export async function publishCluster( const now = new Date().toISOString(); const article = articles.insertArticle({ - title: deriveTitle(body), + title, body, heroImage, video, @@ -461,7 +455,7 @@ export async function publishEventRecap( event: TrackedEvent, constituents: MergedArticle[] ): Promise { - const { body, tagLabels } = await synthesizeRecap(provider, settings.selectedModels.synthesis, event.name, constituents, settings); + const { title, body, tagLabels } = await synthesizeRecap(provider, settings.selectedModels.synthesis, event.name, constituents, settings); const resolvedTags = []; for (const label of tagLabels) { @@ -478,7 +472,7 @@ export async function publishEventRecap( const now = new Date().toISOString(); return articles.insertArticle({ - title: `${event.name}: recap`, + title: title || `${event.name}: recap`, body, heroImage, video: null, diff --git a/backend/src/pipeline/synthesis.ts b/backend/src/pipeline/synthesis.ts index a3be66c..3c04413 100644 --- a/backend/src/pipeline/synthesis.ts +++ b/backend/src/pipeline/synthesis.ts @@ -3,8 +3,17 @@ import type { ContentItem, GlobalSettings, MergedArticle } from '../storage/db/t import { DEFAULT_NUM_CTX, DEFAULT_NUM_PREDICT } from '../inference/ollama-provider.js'; import { logger } from '../storage/db/logs.js'; +const TITLE_DELIMITER = '---TITLE---'; const TAG_DELIMITER = '---TAGS---'; +// Small/quantized models don't always reproduce a literal delimiter exactly — extra +// dashes, an inserted blank line, different case (seen in production with the tag +// delimiter: "---\n\nTAGS---" instead of "---TAGS---", which an exact-string split +// missed entirely, leaking the raw delimiter text into the published body). Splitting +// on a loose regex instead tolerates that variance. +const TITLE_DELIMITER_RE = /-{2,}\s*TITLE\s*-{2,}/i; +const TAG_DELIMITER_RE = /-{2,}\s*TAGS\s*-{2,}/i; + // Ollama truncates prompts that don't fit its context window by keeping a small prefix // and dropping everything else in the middle — silently, with no error, and with no // regard for which sources end up cut (see ollama-provider.ts for the incident that @@ -23,21 +32,25 @@ function capEntryText(text: string, budgetChars: number): string { return text.length > budgetChars ? text.slice(0, budgetChars) + '…' : text; } -const RECAP_SYSTEM_PROMPT_BASE = `You are a neutral news synthesis assistant. Given a chronological list of articles already published about an ongoing tracked event, write a single recap article that: -- Summarizes what has happened across the period covered, in chronological order -- Highlights the most significant developments rather than restating every article -- Stays neutral and factual, without editorializing -- Is 3-5 short paragraphs +const RECAP_SYSTEM_PROMPT_BASE = `You are a neutral news synthesis assistant. Given a chronological list of articles already published about an ongoing tracked event, write your response in exactly three parts, in this order: -After the recap, on a new line, write exactly "${TAG_DELIMITER}" followed by 2-4 short comma-separated topic/entity tags (e.g. proper nouns, named events) that this recap is about. If nothing salient qualifies, leave the tag line empty.`; +1. A short, specific headline for this recap (a single line, ideally under 12 words, no surrounding quotation marks, no trailing period). +2. On a new line, write exactly "${TITLE_DELIMITER}", then the recap article: + - Summarizes what has happened across the period covered, in chronological order + - Highlights the most significant developments rather than restating every article + - Stays neutral and factual, without editorializing + - Is 3-5 short paragraphs +3. On a new line after the recap, write exactly "${TAG_DELIMITER}" followed by 2-4 short comma-separated topic/entity tags (e.g. proper nouns, named events) that this recap is about. If nothing salient qualifies, leave the tag line empty.`; -const SYSTEM_PROMPT_BASE = `You are a neutral news synthesis assistant. Given summaries from multiple news sources describing the same event, write a single original article that: -- Attributes specific claims to the outlet that reported them, using each source's exact name as given below (e.g. if a source is labeled "Source 1 (Reuters)", write "Reuters reported..."). Never invent, guess, or substitute an outlet name that isn't one of the source names actually given below. -- Does not copy phrasing verbatim from any source -- Stays neutral and factual, without editorializing -- Is 2-4 short paragraphs +const SYSTEM_PROMPT_BASE = `You are a neutral news synthesis assistant. Given summaries from multiple news sources describing the same event, write your response in exactly three parts, in this order: -After the article, on a new line, write exactly "${TAG_DELIMITER}" followed by 2-4 short comma-separated topic/entity tags (e.g. proper nouns, named events) that this article is about. If nothing salient qualifies, leave the tag line empty.`; +1. A short, specific headline for this story (a single line, ideally under 12 words, no surrounding quotation marks, no trailing period, no site/outlet name). +2. On a new line, write exactly "${TITLE_DELIMITER}", then the article: + - Attributes specific claims to the outlet that reported them, using each source's exact name as given below (e.g. if a source is labeled "Source 1 (Reuters)", write "Reuters reported..."). Never invent, guess, or substitute an outlet name that isn't one of the source names actually given below. + - Does not copy phrasing verbatim from any source + - Stays neutral and factual, without editorializing + - Is 2-4 short paragraphs +3. On a new line after the article, write exactly "${TAG_DELIMITER}" followed by 2-4 short comma-separated topic/entity tags (e.g. proper nouns, named events) that this article is about. If nothing salient qualifies, leave the tag line empty.`; // Admin-selectable presets (Merge tab, "Writing style") — appended to whichever base // prompt applies. 'default' adds nothing: the base prompts above already describe the @@ -58,10 +71,17 @@ function styleAddendum(settings: GlobalSettings): string { } export interface SynthesisResult { + title: string; body: string; tagLabels: string[]; } +/** Only used when the model doesn't follow the requested title/delimiter format at all — a real headline beats a truncated sentence fragment, but publishing with no title at all is worse than either. */ +function fallbackTitle(body: string): string { + const firstLine = body.split('\n')[0]; + return firstLine.length > 100 ? firstLine.slice(0, 97) + '…' : firstLine; +} + function buildPrompt(items: ContentItem[], sourceNames: Map): string { const budgetPerItem = Math.max(MIN_ENTRY_CHARS, Math.floor(MAX_INPUT_CHARS / items.length)); let truncated = 0; @@ -88,13 +108,24 @@ function buildPrompt(items: ContentItem[], sourceNames: Map): st } function parseResult(raw: string): SynthesisResult { - const [body, tagSection] = raw.split(TAG_DELIMITER); + const [beforeTags, tagSection] = raw.split(TAG_DELIMITER_RE); const tagLabels = (tagSection ?? '') .split(',') .map((t) => t.trim()) .filter((t) => t.length > 0 && t.length < 60); - return { body: body.trim(), tagLabels }; + const titleSplit = (beforeTags ?? raw).split(TITLE_DELIMITER_RE); + const titlePart = titleSplit[0]; + // join() rather than titleSplit[1] in case the delimiter text somehow appears again + // inside the body itself — keeps that content rather than silently dropping it. + const bodyPart = titleSplit.length > 1 ? titleSplit.slice(1).join('') : undefined; + // If the title delimiter never showed up, the model didn't follow the requested + // format — treat the whole thing as body rather than mistaking the article itself + // for a "title", and fall back to the old truncated-first-line heuristic. + const body = (bodyPart ?? titlePart).trim(); + const title = bodyPart !== undefined ? titlePart.trim() : fallbackTitle(body); + + return { title, body, tagLabels }; } export async function synthesizeArticle(