Erster Produktionslauf schlug mit HTTP 500 fehl, lokal lief derselbe Code grün. Ursache: Vercel kompiliert jede Datei einzeln statt sie zu bündeln, und weil das Projekt `"type": "module"` ist, verlangt Nodes ESM-Lader explizite Dateiendungen in relativen Importen. Vite löst im Entwicklungsbetrieb grosszügiger auf und verdeckt das. Das ist genau die Klasse Fehler, die kein Typecheck und kein Test findet — nur ein echter Aufruf gegen die deployte Funktion. Geprüft gegen Produktion: HTTP 200, 20 Einträge aus 3/3 Quellen in 24 Sekunden, Beurteilung läuft, Originalquellen verlinkt, keine Laufzeitfehler. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
275 lines
9.6 KiB
TypeScript
275 lines
9.6 KiB
TypeScript
/**
|
||
* Livia Research-PoC — die drei realen Quellen.
|
||
*
|
||
* Bewusst drei einzelne Funktionen statt einer Connector-Abstraktion: es sind
|
||
* drei Quellen, und jede liest anders. Eine gemeinsame Schnittstelle würde
|
||
* hier nur die Unterschiede verstecken, die man beim Debuggen braucht.
|
||
*
|
||
* Kein Browser, kein Playwright, kein Scheduler — nur `fetch()` und ein
|
||
* HTML-Parser. Fällt eine Quelle aus, liefert sie einen Fehler und die anderen
|
||
* beiden laufen weiter (§21).
|
||
*/
|
||
|
||
import * as cheerio from 'cheerio'
|
||
import { RESEARCH_SOURCES } from '../../src/lib/researchSources.js'
|
||
|
||
/** Gemeinsames Minimalformat aller Quellen (§9). */
|
||
export interface ResearchItem {
|
||
source: string
|
||
title: string
|
||
url: string
|
||
text: string
|
||
publishedAt?: string
|
||
}
|
||
|
||
/**
|
||
* Die Adressen kommen aus `src/lib/researchSources.ts` — derselben Liste, die
|
||
* die Oberfläche unter «Angebundene Kanäle & Systeme» anzeigt. Damit kann der
|
||
* angezeigte Link nicht von dem abweichen, was tatsächlich abgerufen wird.
|
||
*/
|
||
export const SOURCES = RESEARCH_SOURCES
|
||
|
||
/** Höchstzahl Artikel je Newsquelle — hält Laufzeit und Tokenverbrauch klein (§22). */
|
||
const MAX_ARTICLES = 10
|
||
const FETCH_TIMEOUT_MS = 15_000
|
||
const UA = 'Mozilla/5.0 (compatible; PropertyMatch-LiviaResearch/1.0; +https://property-match-virid.vercel.app)'
|
||
|
||
async function getText(url: string, extraHeaders: Record<string, string> = {}): Promise<string> {
|
||
const ctrl = new AbortController()
|
||
const timer = setTimeout(() => ctrl.abort(), FETCH_TIMEOUT_MS)
|
||
try {
|
||
const res = await fetch(url, {
|
||
headers: { 'User-Agent': UA, 'Accept-Language': 'de-CH,de;q=0.9', ...extraHeaders },
|
||
redirect: 'follow',
|
||
signal: ctrl.signal,
|
||
})
|
||
if (!res.ok) throw new Error(`HTTP ${res.status}`)
|
||
return await res.text()
|
||
} finally {
|
||
clearTimeout(timer)
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Container, in denen der Artikeltext bevorzugt gesucht wird — von eng nach weit.
|
||
*
|
||
* Die Reihenfolge ist der Unterschied zwischen brauchbarem und unbrauchbarem
|
||
* Input: Nimmt man gleich alle `<p>` der Seite, stehen bei der Handelskammer
|
||
* zuerst Telefonnummern, Öffnungszeiten und Spam-geschützte Mailadressen im
|
||
* Text — und das Modell beurteilt dann die Fusszeile mit.
|
||
*/
|
||
const ARTICLE_CONTAINERS = ['.news-single', '.news-detail', 'article', 'main']
|
||
|
||
/** Zeilen, die erkennbar Kontaktangaben statt Inhalt sind. */
|
||
const BOILERPLATE = /(dont-like-spam|ich-will-kein-spam|^\+41|Montag\s*[–-]\s*Freitag)/i
|
||
|
||
/**
|
||
* Haupttext einer Artikelseite.
|
||
*
|
||
* Absichtlich grob: Skripte und Navigation raus, Absätze aus dem engsten
|
||
* passenden Container einsammeln, kürzen. Für die Beurteilung genügen die
|
||
* ersten Absätze — ein vollwertiger Lesemodus wäre für einen
|
||
* Machbarkeitsnachweis Aufwand ohne zusätzliche Aussage.
|
||
*/
|
||
function extractArticleText(html: string): string {
|
||
const $ = cheerio.load(html)
|
||
$('script, style, nav, header, footer, noscript, form').remove()
|
||
|
||
const sammle = (selector: string): string[] => {
|
||
const parts: string[] = []
|
||
$(selector).each((_, el) => {
|
||
const t = $(el).text().replace(/\s+/g, ' ').trim()
|
||
if (t.length > 40 && !BOILERPLATE.test(t)) parts.push(t)
|
||
})
|
||
return parts
|
||
}
|
||
|
||
for (const container of ARTICLE_CONTAINERS) {
|
||
if ($(container).length === 0) continue
|
||
const parts = sammle(`${container} p`)
|
||
const text = parts.join('\n')
|
||
if (text.length > 200) return text.slice(0, 2500)
|
||
}
|
||
|
||
return sammle('p').join('\n').slice(0, 2500)
|
||
}
|
||
|
||
/**
|
||
* Der Linktext trägt bei der Handelskammer das Publikationsdatum vorneweg
|
||
* («28. August 2026 Bystronic übernimmt …») und hinten ein »-Zeichen. Beides
|
||
* gehört nicht in den Titel, den das Modell zu lesen bekommt.
|
||
*/
|
||
function cleanTitle(raw: string): string {
|
||
return raw
|
||
.replace(/^\d{1,2}\.\s*\p{L}+\s+\d{4}\s*/u, '')
|
||
.replace(/\s*[»›>]\s*$/, '')
|
||
.trim()
|
||
}
|
||
|
||
function absolute(href: string, base: string): string {
|
||
try {
|
||
return new URL(href, base).toString()
|
||
} catch {
|
||
return href
|
||
}
|
||
}
|
||
|
||
/** Quelle A — Zürcher Handelskammer, Unternehmensmeldungen aus der Region (§4). */
|
||
export async function fetchZhkResearch(): Promise<ResearchItem[]> {
|
||
const listHtml = await getText(SOURCES.ZHK.url)
|
||
const $ = cheerio.load(listHtml)
|
||
|
||
const links: { url: string; title: string }[] = []
|
||
const seen = new Set<string>()
|
||
$('a[href*="/de/wirtschaft-und-politik/news/"]').each((_, el) => {
|
||
const href = $(el).attr('href')
|
||
if (!href) return
|
||
const url = absolute(href, SOURCES.ZHK.url)
|
||
if (seen.has(url)) return
|
||
seen.add(url)
|
||
const title = $(el).text().replace(/\s+/g, ' ').trim()
|
||
links.push({ url, title })
|
||
})
|
||
|
||
const chosen = links.slice(0, MAX_ARTICLES)
|
||
const items: ResearchItem[] = []
|
||
for (const l of chosen) {
|
||
try {
|
||
const html = await getText(l.url)
|
||
const $a = cheerio.load(html)
|
||
const title = cleanTitle(l.title) || $a('h1').first().text().replace(/\s+/g, ' ').trim()
|
||
const published =
|
||
$a('time').first().attr('datetime') ||
|
||
$a('meta[property="article:published_time"]').attr('content') ||
|
||
undefined
|
||
items.push({
|
||
source: SOURCES.ZHK.name,
|
||
title,
|
||
url: l.url,
|
||
text: extractArticleText(html),
|
||
publishedAt: published,
|
||
})
|
||
} catch {
|
||
// Ein einzelner Artikel darf den Lauf nicht kippen.
|
||
}
|
||
}
|
||
return items
|
||
}
|
||
|
||
/** Quelle B — Greater Zurich Area, Ansiedlungen und Markteintritte (§5). */
|
||
export async function fetchGreaterZurichResearch(): Promise<ResearchItem[]> {
|
||
const listHtml = await getText(SOURCES.GZA.url)
|
||
const $ = cheerio.load(listHtml)
|
||
|
||
const links: { url: string; title: string }[] = []
|
||
const seen = new Set<string>()
|
||
$('a[href*="/de/news/"]').each((_, el) => {
|
||
const href = $(el).attr('href')
|
||
if (!href) return
|
||
const url = absolute(href, SOURCES.GZA.url)
|
||
if (seen.has(url) || /\/de\/news\/?$/.test(url)) return
|
||
seen.add(url)
|
||
links.push({ url, title: $(el).text().replace(/\s+/g, ' ').trim() })
|
||
})
|
||
|
||
const chosen = links.slice(0, MAX_ARTICLES)
|
||
const items: ResearchItem[] = []
|
||
for (const l of chosen) {
|
||
try {
|
||
const html = await getText(l.url)
|
||
const $a = cheerio.load(html)
|
||
const title = cleanTitle(l.title) || $a('h1').first().text().replace(/\s+/g, ' ').trim()
|
||
items.push({
|
||
source: SOURCES.GZA.name,
|
||
title,
|
||
url: l.url,
|
||
text: extractArticleText(html),
|
||
publishedAt:
|
||
$a('time').first().attr('datetime') ||
|
||
$a('meta[property="article:published_time"]').attr('content') ||
|
||
undefined,
|
||
})
|
||
} catch {
|
||
// siehe oben
|
||
}
|
||
}
|
||
return items
|
||
}
|
||
|
||
/**
|
||
* Quelle C — Zefix / Open Data Kanton Zürich, direkt als CSV (§6).
|
||
*
|
||
* **Wichtige Feststellung zur Datenlage:** Die Datei enthält keine Firmennamen.
|
||
* Sie führt Tageszahlen von Neugründungen je NOGA-Branche für den Kanton
|
||
* Zürich — Spalten `br_abschnitt_desc, br_abschnitt_code, wirtschaftssektor_desc,
|
||
* wirtschaftssektor_code, location, date, value`.
|
||
*
|
||
* Daraus lässt sich deshalb kein Unternehmenslead bauen, sondern nur ein
|
||
* aggregiertes Frühsignal zum Zielmarkt. Genau so wird es weitergegeben.
|
||
* Firmennamen zu erfinden, um das erwartete Format zu treffen, wäre die eine
|
||
* Sache, die Livia nie tun darf.
|
||
*/
|
||
export async function fetchZefixResearch(): Promise<ResearchItem[]> {
|
||
const csv = await getText(SOURCES.ZEFIX.url)
|
||
const lines = csv.split(/\r?\n/)
|
||
if (lines.length < 2) throw new Error('CSV enthält keine Daten')
|
||
|
||
// Die Beschreibung kann Kommas enthalten — die sechs Pflichtfelder stehen
|
||
// deshalb von hinten, der Rest davor ist die Branchenbezeichnung.
|
||
interface Row { branche: string; datum: string; anzahl: number }
|
||
const rows: Row[] = []
|
||
let neuestesDatum = ''
|
||
|
||
for (let i = 1; i < lines.length; i++) {
|
||
const line = lines[i]
|
||
if (!line) continue
|
||
const parts = line.split(',')
|
||
if (parts.length < 7) continue
|
||
const anzahl = Number(parts[parts.length - 1])
|
||
const datum = parts[parts.length - 2]
|
||
const ort = parts[parts.length - 3]
|
||
const branche = parts.slice(0, parts.length - 6).join(',').trim()
|
||
if (ort !== 'ZH' || !Number.isFinite(anzahl)) continue
|
||
if (datum > neuestesDatum) neuestesDatum = datum
|
||
rows.push({ branche, datum, anzahl })
|
||
}
|
||
|
||
if (!neuestesDatum) throw new Error('Keine Zürcher Datenzeilen gefunden')
|
||
|
||
// Fenster: die letzten 30 Tage vor dem jüngsten Datum im Bestand.
|
||
const bis = new Date(neuestesDatum)
|
||
const von = new Date(bis.getTime() - 30 * 24 * 60 * 60 * 1000)
|
||
const vonIso = von.toISOString().slice(0, 10)
|
||
|
||
const jeBranche = new Map<string, number>()
|
||
let gesamt = 0
|
||
for (const r of rows) {
|
||
if (r.datum < vonIso || r.datum > neuestesDatum) continue
|
||
if (r.branche === 'Alle Branchen') continue
|
||
if (r.anzahl <= 0) continue
|
||
jeBranche.set(r.branche, (jeBranche.get(r.branche) ?? 0) + r.anzahl)
|
||
gesamt += r.anzahl
|
||
}
|
||
|
||
const top = [...jeBranche.entries()].sort((a, b) => b[1] - a[1]).slice(0, 8)
|
||
if (top.length === 0) throw new Error('Keine Neugründungen im Auswertungsfenster')
|
||
|
||
const text = [
|
||
`Amtliche Statistik des Kantons Zürich zu Handelsregister-Neugründungen.`,
|
||
`Auswertungsfenster ${vonIso} bis ${neuestesDatum}: insgesamt ${gesamt} Neugründungen im Kanton Zürich.`,
|
||
`Verteilung nach Branche:`,
|
||
...top.map(([b, n]) => `- ${b}: ${n}`),
|
||
``,
|
||
`Hinweis zur Datenart: Der Datensatz enthält ausschliesslich Tageszahlen je Branche.`,
|
||
`Firmennamen, Adressen und Gemeinden sind darin nicht enthalten.`,
|
||
].join('\n')
|
||
|
||
return [{
|
||
source: SOURCES.ZEFIX.name,
|
||
title: `Kanton Zürich: ${gesamt} Handelsregister-Neugründungen in 30 Tagen`,
|
||
url: SOURCES.ZEFIX.url,
|
||
text,
|
||
publishedAt: neuestesDatum,
|
||
}]
|
||
}
|