feat: extract articles with trafilatura on /load

Run trafilatura server-side on the rendered DOM (page.content()), so
JS-rendered pages stay fully visible to the extractor; fall back to
innerText when trafilatura cannot score any main content.
This commit is contained in:
ThePhaseless
2026-08-08 01:26:35 +02:00
parent cd4359a1dc
commit 0dff659e34
4 changed files with 298 additions and 39 deletions
+5 -1
View File
@@ -5,6 +5,7 @@ from __future__ import annotations
from hmac import compare_digest
from typing import Annotated
import trafilatura
from fastapi import APIRouter, Depends, Header, HTTPException
from playwright.async_api import Page
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
@@ -38,7 +39,10 @@ def require_auth(authorization: Annotated[str | None, Header()] = None) -> None:
async def _extract_content(page: Page) -> str:
"""Return the page's visible text, with blank lines removed."""
"""Return the page's main article text, falling back to visible text."""
article = trafilatura.extract(await page.content())
if article:
return article
result = await page.evaluate("() => document.body ? document.body.innerText : ''")
return "\n".join(line.strip() for line in result.splitlines() if line.strip())