feat(livia): drei Übersichtsseiten je Newsquelle statt einer
Eine Seite reicht nicht. Die Handelskammer erneuert ihre obersten zehn Meldungen im Lauf eines Tages; stehen dort gerade Personalien und Meinungsbeiträge, findet Livia zu Recht nichts, und die Demo wirkt schwach, obwohl die Mechanik stimmt. Genau das war zwischen zwei Läufen zu beobachten. Neu werden je Quelle drei Übersichtsseiten gelesen — bei der Handelskammer über `…/page/N.html`, bei Greater Zurich über `?page=N` —, gedeckelt auf 24 Artikel je Quelle. Die Artikel werden parallel geholt statt nacheinander, die Beurteilung läuft in begrenzt parallelen Fünferstapeln. Wirkung, gemessen am selben Tag: aus 20 gelesenen Einträgen und 1 Lead werden 49 Einträge und 8 Leads, darunter zwei mit hoher Relevanz — eine Arealentwicklung über 38 000 m² und eine Standorteröffnung. Laufzeit 30 Sekunden statt 16, weiterhin klar innerhalb des 60-Sekunden-Limits. Die Begrenzung der Gleichzeitigkeit ist kein Detail: alle Teilstapel auf einmal loszuschicken wäre schneller, läuft aber in die Drosselung des Anbieters — und ein gedrosselter Lauf sieht für den Nutzer aus wie ein kaputter. Ein gescheiterter Teilstapel reisst die übrigen nicht mehr mit. Geprüft: tsc 0 Fehler, eslint 0 Fehler/0 Warnungen, Build erfolgreich, Live-Lauf 49 Einträge aus 3/3 Quellen. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
1 parent
f53f3be1e8
commit
f58d841e7d
2 files changed
+142
-69
No files matched your search
@@ -90,6 +90,15 @@ function extractJson(text: string): unknown {
|
|||||||
*/
|
*/
|
||||||
const BATCH_SIZE = 5
|
const BATCH_SIZE = 5
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Wie viele Teilstapel gleichzeitig beurteilt werden.
|
||||||
|
*
|
||||||
|
* Alles auf einmal loszuschicken wäre schneller, läuft aber in die Drosselung
|
||||||
|
* des Anbieters — und ein gedrosselter Lauf sieht für den Nutzer aus wie ein
|
||||||
|
* kaputter.
|
||||||
|
*/
|
||||||
|
const ANALYSIS_CONCURRENCY = 5
|
||||||
|
|
||||||
function chunk<T>(list: T[], size: number): T[][] {
|
function chunk<T>(list: T[], size: number): T[][] {
|
||||||
const out: T[][] = []
|
const out: T[][] = []
|
||||||
for (let i = 0; i < list.length; i += size) out.push(list.slice(i, i + size))
|
for (let i = 0; i < list.length; i += size) out.push(list.slice(i, i + size))
|
||||||
@@ -148,11 +157,28 @@ export async function analyzeResearchItems(items: ResearchItem[]): Promise<Analy
|
|||||||
|
|
||||||
const client = new Anthropic({ apiKey })
|
const client = new Anthropic({ apiKey })
|
||||||
|
|
||||||
// Parallel: die Laufzeit ist damit die des langsamsten Teilstapels und nicht
|
// Parallel, aber begrenzt: die Laufzeit ist damit nicht die Summe der
|
||||||
// deren Summe — das hält den Aufruf innerhalb des Zeitlimits der Funktion.
|
// Teilstapel, und gleichzeitig laufen nie so viele Anfragen, dass der
|
||||||
const batches = await Promise.all(
|
// Anbieter drosselt.
|
||||||
chunk(items, BATCH_SIZE).map(b => analyzeBatch(client, b)),
|
const teilstapel = chunk(items, BATCH_SIZE)
|
||||||
|
const ergebnisse: Awaited<ReturnType<typeof analyzeBatch>>[] = new Array(teilstapel.length)
|
||||||
|
let next = 0
|
||||||
|
const worker = async (): Promise<void> => {
|
||||||
|
for (;;) {
|
||||||
|
const i = next++
|
||||||
|
if (i >= teilstapel.length) return
|
||||||
|
// Ein gescheiterter Teilstapel darf die übrigen nicht mitreissen.
|
||||||
|
try {
|
||||||
|
ergebnisse[i] = await analyzeBatch(client, teilstapel[i])
|
||||||
|
} catch {
|
||||||
|
ergebnisse[i] = []
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
await Promise.all(
|
||||||
|
Array.from({ length: Math.min(ANALYSIS_CONCURRENCY, teilstapel.length) }, worker),
|
||||||
)
|
)
|
||||||
|
const batches = ergebnisse
|
||||||
|
|
||||||
const leads: ResearchLead[] = []
|
const leads: ResearchLead[] = []
|
||||||
let discarded = 0
|
let discarded = 0
|
||||||
|
|||||||
+112
-65
@@ -29,8 +29,23 @@ export interface ResearchItem {
|
|||||||
*/
|
*/
|
||||||
export const SOURCES = RESEARCH_SOURCES
|
export const SOURCES = RESEARCH_SOURCES
|
||||||
|
|
||||||
/** Höchstzahl Artikel je Newsquelle — hält Laufzeit und Tokenverbrauch klein (§22). */
|
/**
|
||||||
const MAX_ARTICLES = 10
|
* Wie viele Übersichtsseiten je Newsquelle gelesen werden.
|
||||||
|
*
|
||||||
|
* Eine Seite allein reicht nicht: Die Handelskammer erneuert ihre obersten
|
||||||
|
* zehn Meldungen im Lauf eines Tages, und wenn dort gerade Personalien und
|
||||||
|
* Meinungsbeiträge stehen, findet Livia zu Recht nichts. Drei Seiten decken
|
||||||
|
* bei der Handelskammer rund 40 und bei Greater Zurich rund 27 Meldungen ab —
|
||||||
|
* damit hängt das Ergebnis nicht mehr am Zufall eines einzelnen Tages.
|
||||||
|
*/
|
||||||
|
const LIST_PAGES = 3
|
||||||
|
|
||||||
|
/** Höchstzahl Artikel je Newsquelle — hält Laufzeit und Tokenverbrauch im Rahmen (§22). */
|
||||||
|
const MAX_ARTICLES = 24
|
||||||
|
|
||||||
|
/** Gleichzeitige Artikelabrufe. Nacheinander wären 24 Abrufe je Quelle zu langsam. */
|
||||||
|
const FETCH_CONCURRENCY = 8
|
||||||
|
|
||||||
const FETCH_TIMEOUT_MS = 15_000
|
const FETCH_TIMEOUT_MS = 15_000
|
||||||
const UA = 'Mozilla/5.0 (compatible; PropertyMatch-LiviaResearch/1.0; +https://property-match-virid.vercel.app)'
|
const UA = 'Mozilla/5.0 (compatible; PropertyMatch-LiviaResearch/1.0; +https://property-match-virid.vercel.app)'
|
||||||
|
|
||||||
@@ -114,86 +129,118 @@ function absolute(href: string, base: string): string {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Quelle A — Zürcher Handelskammer, Unternehmensmeldungen aus der Region (§4). */
|
/** Kartiert mit begrenzter Gleichzeitigkeit — schneller als nacheinander, ohne die Quelle zu fluten. */
|
||||||
export async function fetchZhkResearch(): Promise<ResearchItem[]> {
|
async function mapLimit<T, R>(items: T[], limit: number, fn: (item: T) => Promise<R>): Promise<R[]> {
|
||||||
const listHtml = await getText(SOURCES.ZHK.url)
|
const out: R[] = new Array<R>(items.length)
|
||||||
const $ = cheerio.load(listHtml)
|
let next = 0
|
||||||
|
const worker = async (): Promise<void> => {
|
||||||
const links: { url: string; title: string }[] = []
|
for (;;) {
|
||||||
const seen = new Set<string>()
|
const i = next++
|
||||||
$('a[href*="/de/wirtschaft-und-politik/news/"]').each((_, el) => {
|
if (i >= items.length) return
|
||||||
const href = $(el).attr('href')
|
out[i] = await fn(items[i])
|
||||||
if (!href) return
|
|
||||||
const url = absolute(href, SOURCES.ZHK.url)
|
|
||||||
if (seen.has(url)) return
|
|
||||||
seen.add(url)
|
|
||||||
const title = $(el).text().replace(/\s+/g, ' ').trim()
|
|
||||||
links.push({ url, title })
|
|
||||||
})
|
|
||||||
|
|
||||||
const chosen = links.slice(0, MAX_ARTICLES)
|
|
||||||
const items: ResearchItem[] = []
|
|
||||||
for (const l of chosen) {
|
|
||||||
try {
|
|
||||||
const html = await getText(l.url)
|
|
||||||
const $a = cheerio.load(html)
|
|
||||||
const title = cleanTitle(l.title) || $a('h1').first().text().replace(/\s+/g, ' ').trim()
|
|
||||||
const published =
|
|
||||||
$a('time').first().attr('datetime') ||
|
|
||||||
$a('meta[property="article:published_time"]').attr('content') ||
|
|
||||||
undefined
|
|
||||||
items.push({
|
|
||||||
source: SOURCES.ZHK.name,
|
|
||||||
title,
|
|
||||||
url: l.url,
|
|
||||||
text: extractArticleText(html),
|
|
||||||
publishedAt: published,
|
|
||||||
})
|
|
||||||
} catch {
|
|
||||||
// Ein einzelner Artikel darf den Lauf nicht kippen.
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return items
|
await Promise.all(Array.from({ length: Math.min(limit, items.length) }, worker))
|
||||||
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Quelle B — Greater Zurich Area, Ansiedlungen und Markteintritte (§5). */
|
interface Artikellink { url: string; title: string }
|
||||||
export async function fetchGreaterZurichResearch(): Promise<ResearchItem[]> {
|
|
||||||
const listHtml = await getText(SOURCES.GZA.url)
|
|
||||||
const $ = cheerio.load(listHtml)
|
|
||||||
|
|
||||||
const links: { url: string; title: string }[] = []
|
/**
|
||||||
const seen = new Set<string>()
|
* Artikel zu Einträgen machen — parallel, und ein einzelner Ausfall kippt den
|
||||||
$('a[href*="/de/news/"]').each((_, el) => {
|
* Lauf nicht. Statt einer Ausnahme entsteht dann schlicht kein Eintrag.
|
||||||
const href = $(el).attr('href')
|
*/
|
||||||
if (!href) return
|
async function ladeArtikel(links: Artikellink[], quelle: string): Promise<ResearchItem[]> {
|
||||||
const url = absolute(href, SOURCES.GZA.url)
|
const items = await mapLimit(links, FETCH_CONCURRENCY, async (l): Promise<ResearchItem | null> => {
|
||||||
if (seen.has(url) || /\/de\/news\/?$/.test(url)) return
|
|
||||||
seen.add(url)
|
|
||||||
links.push({ url, title: $(el).text().replace(/\s+/g, ' ').trim() })
|
|
||||||
})
|
|
||||||
|
|
||||||
const chosen = links.slice(0, MAX_ARTICLES)
|
|
||||||
const items: ResearchItem[] = []
|
|
||||||
for (const l of chosen) {
|
|
||||||
try {
|
try {
|
||||||
const html = await getText(l.url)
|
const html = await getText(l.url)
|
||||||
const $a = cheerio.load(html)
|
const $a = cheerio.load(html)
|
||||||
const title = cleanTitle(l.title) || $a('h1').first().text().replace(/\s+/g, ' ').trim()
|
return {
|
||||||
items.push({
|
source: quelle,
|
||||||
source: SOURCES.GZA.name,
|
title: cleanTitle(l.title) || $a('h1').first().text().replace(/\s+/g, ' ').trim(),
|
||||||
title,
|
|
||||||
url: l.url,
|
url: l.url,
|
||||||
text: extractArticleText(html),
|
text: extractArticleText(html),
|
||||||
publishedAt:
|
publishedAt:
|
||||||
$a('time').first().attr('datetime') ||
|
$a('time').first().attr('datetime') ||
|
||||||
$a('meta[property="article:published_time"]').attr('content') ||
|
$a('meta[property="article:published_time"]').attr('content') ||
|
||||||
undefined,
|
undefined,
|
||||||
})
|
}
|
||||||
} catch {
|
} catch {
|
||||||
// siehe oben
|
return null
|
||||||
}
|
}
|
||||||
|
})
|
||||||
|
return items.filter((i): i is ResearchItem => i !== null && i.text.length > 120)
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Artikellinks aus mehreren Übersichtsseiten einsammeln.
|
||||||
|
*
|
||||||
|
* Die Listenseiten werden parallel geholt, das Ergebnis aber in Seitenreihenfolge
|
||||||
|
* zusammengesetzt: Seite 1 trägt die neuesten Meldungen, und die sollen zuerst
|
||||||
|
* kommen, wenn die Obergrenze greift.
|
||||||
|
*/
|
||||||
|
async function sammleLinks(
|
||||||
|
seitenUrls: string[],
|
||||||
|
basis: string,
|
||||||
|
selektor: string,
|
||||||
|
ausschluss?: RegExp,
|
||||||
|
): Promise<Artikellink[]> {
|
||||||
|
const seiten = await Promise.all(
|
||||||
|
seitenUrls.map(async u => {
|
||||||
|
try {
|
||||||
|
return await getText(u)
|
||||||
|
} catch {
|
||||||
|
return ''
|
||||||
|
}
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
|
||||||
|
const links: Artikellink[] = []
|
||||||
|
const seen = new Set<string>()
|
||||||
|
for (const html of seiten) {
|
||||||
|
if (!html) continue
|
||||||
|
const $ = cheerio.load(html)
|
||||||
|
$(selektor).each((_, el) => {
|
||||||
|
const href = $(el).attr('href')
|
||||||
|
if (!href) return
|
||||||
|
const url = absolute(href, basis)
|
||||||
|
if (seen.has(url)) return
|
||||||
|
if (ausschluss?.test(url)) return
|
||||||
|
seen.add(url)
|
||||||
|
links.push({ url, title: $(el).text().replace(/\s+/g, ' ').trim() })
|
||||||
|
})
|
||||||
}
|
}
|
||||||
return items
|
return links.slice(0, MAX_ARTICLES)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Quelle A — Zürcher Handelskammer, Unternehmensmeldungen aus der Region (§4). */
|
||||||
|
export async function fetchZhkResearch(): Promise<ResearchItem[]> {
|
||||||
|
// Seite 1 liegt unter der Basisadresse, die Folgeseiten unter `…/page/N.html`.
|
||||||
|
const seiten = [
|
||||||
|
SOURCES.ZHK.url,
|
||||||
|
...Array.from({ length: LIST_PAGES - 1 }, (_, i) =>
|
||||||
|
`https://www.zhk.ch/de/wirtschaft-und-politik/news-liste/page/${i + 2}.html`),
|
||||||
|
]
|
||||||
|
const links = await sammleLinks(
|
||||||
|
seiten,
|
||||||
|
SOURCES.ZHK.url,
|
||||||
|
'a[href*="/de/wirtschaft-und-politik/news/"]',
|
||||||
|
)
|
||||||
|
return ladeArtikel(links, SOURCES.ZHK.name)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Quelle B — Greater Zurich Area, Ansiedlungen und Markteintritte (§5). */
|
||||||
|
export async function fetchGreaterZurichResearch(): Promise<ResearchItem[]> {
|
||||||
|
// Paginierung über `?page=N`, beginnend bei 0.
|
||||||
|
const seiten = Array.from({ length: LIST_PAGES }, (_, i) =>
|
||||||
|
i === 0 ? SOURCES.GZA.url : `${SOURCES.GZA.url}?page=${i}`)
|
||||||
|
const links = await sammleLinks(
|
||||||
|
seiten,
|
||||||
|
SOURCES.GZA.url,
|
||||||
|
'a[href*="/de/news/"]',
|
||||||
|
/\/de\/news\/?$/,
|
||||||
|
)
|
||||||
|
return ladeArtikel(links, SOURCES.GZA.name)
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
Reference in new issue
Block a user