mirror of
https://github.com/ThePhaseless/Byparr.git
synced 2026-10-06 06:05:24 +01:00
feat: extract articles with trafilatura on /load
Run trafilatura server-side on the rendered DOM (page.content()), so JS-rendered pages stay fully visible to the extractor; fall back to innerText when trafilatura cannot score any main content.
This commit is contained in:
+5
-1
@@ -5,6 +5,7 @@ from __future__ import annotations
|
||||
from hmac import compare_digest
|
||||
from typing import Annotated
|
||||
|
||||
import trafilatura
|
||||
from fastapi import APIRouter, Depends, Header, HTTPException
|
||||
from playwright.async_api import Page
|
||||
from playwright.async_api import TimeoutError as PlaywrightTimeoutError
|
||||
@@ -38,7 +39,10 @@ def require_auth(authorization: Annotated[str | None, Header()] = None) -> None:
|
||||
|
||||
|
||||
async def _extract_content(page: Page) -> str:
|
||||
"""Return the page's visible text, with blank lines removed."""
|
||||
"""Return the page's main article text, falling back to visible text."""
|
||||
article = trafilatura.extract(await page.content())
|
||||
if article:
|
||||
return article
|
||||
result = await page.evaluate("() => document.body ? document.body.innerText : ''")
|
||||
return "\n".join(line.strip() for line in result.splitlines() if line.strip())
|
||||
|
||||
|
||||
Reference in New Issue
Block a user