File size: 6,343 Bytes
0e17b96
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
#!/usr/bin/env python3
"""clean-read x402 API — URL → clean Markdown extraction as a paid service (x402 v2).

Deploy: uvicorn scripts.clean_read_api.main:app --host 0.0.0.0 --port $PORT
Local:  uvicorn scripts.clean_read_api.main:app --port 8403

Endpoints:
  GET  /        — Service info (free)
  GET  /health  — Health check (free)
  POST /read    — Fetch URL, strip boilerplate, return main content as Markdown. $0.005/call (x402)

Same stack as scripts/x402_api/main.py (skill-audit): official x402 v2 SDK,
USDC on Base mainnet, Dexter facilitator (zero-gate, 0% seller fee, auto-lists
on Bazaar discovery after first settled payment).

Why this service: agents constantly need "give me the readable text of this page"
(the Jina-reader use case) — extraction via trafilatura, priced at high-volume
$0.005 like tokenguard (our best-selling price point).
"""
import os
from datetime import datetime
from fastapi import FastAPI, HTTPException
from fastapi.middleware.cors import CORSMiddleware
from pydantic import BaseModel
from typing import Optional

WALLET = os.environ.get("BASE_WALLET_ADDRESS", "0x5bCDA55247B238a573A968B234F788a2D35664Dd")
BASE_MAINNET = "eip155:8453"
FACILITATOR_URL = os.environ.get("FACILITATOR_URL", "https://x402.dexter.cash")

app = FastAPI(
    title="clean-read API",
    description="URL to clean Markdown for AI agents. x402 v2 micropayments on Base.",
    version="1.0.0",
)
app.add_middleware(CORSMiddleware, allow_origins=["*"], allow_methods=["*"], allow_headers=["*"])

_x402_available = False
try:
    from x402.http import FacilitatorConfig, HTTPFacilitatorClient, PaymentOption
    from x402.http.middleware.fastapi import PaymentMiddlewareASGI
    from x402.http.types import RouteConfig
    from x402.mechanisms.evm.exact import ExactEvmServerScheme
    from x402.server import x402ResourceServer
    from x402.extensions.bazaar import declare_discovery_extension, OutputConfig

    facilitator = HTTPFacilitatorClient(FacilitatorConfig(url=FACILITATOR_URL))
    server = x402ResourceServer(facilitator)
    server.register(BASE_MAINNET, ExactEvmServerScheme())

    ext = declare_discovery_extension(
        input={"url": "https://example.com/article"},
        input_schema={
            "properties": {
                "url": {"type": "string", "format": "uri", "description": "Page to fetch and clean"},
                "include_links": {"type": "boolean", "description": "Keep hyperlinks in the Markdown (default true)"},
            },
            "required": ["url"],
        },
        body_type="json",
        output=OutputConfig(example={
            "url": "https://example.com/article",
            "title": "Article title",
            "markdown": "# Article title\n\nMain content…",
            "word_count": 1234,
        }),
    )
    ext["bazaar"]["info"]["input"]["method"] = "POST"

    routes = {
        "POST /read": RouteConfig(
            accepts=[PaymentOption(scheme="exact", pay_to=WALLET, price="$0.005", network=BASE_MAINNET)],
            mime_type="application/json",
            description="Fetch a URL and return its main content as clean Markdown (boilerplate stripped)",
            extensions=ext,
        ),
    }
    app.add_middleware(PaymentMiddlewareASGI, routes=routes, server=server)
    _x402_available = True
except Exception as e:  # pragma: no cover
    print(f"  x402 v2 init warning: {type(e).__name__}: {e}")


class ReadRequest(BaseModel):
    url: str
    include_links: Optional[bool] = True
    max_size: Optional[int] = 2_000_000  # 2MB raw HTML cap


@app.get("/")
async def root():
    return {
        "service": "clean-read API",
        "version": "1.0.0",
        "description": "URL → clean Markdown. Boilerplate/nav/ads stripped, main content only. Built for AI agents.",
        "endpoints": {
            "GET /": "Service info (free)",
            "GET /health": "Health check (free)",
            "POST /read": "Fetch URL → Markdown ($0.005 USDC)",
        },
        "payment": {
            "method": "x402",
            "x402_version": 2,
            "currency": "USDC",
            "network": "Base (eip155:8453)",
            "facilitator": FACILITATOR_URL,
            "wallet": WALLET,
            "x402_enabled": _x402_available,
        },
    }


@app.get("/health")
async def health():
    return {"status": "ok", "timestamp": datetime.utcnow().isoformat() + "Z", "x402_enabled": _x402_available}


@app.post("/read")
async def read_url(req: ReadRequest):
    url = req.url
    if not url or not url.startswith(("http://", "https://")):
        raise HTTPException(400, "valid http/https URL required")

    import httpx
    try:
        async with httpx.AsyncClient(follow_redirects=True, timeout=20.0) as client:
            resp = await client.get(url, headers={"User-Agent": "Mozilla/5.0 (compatible; clean-read/1.0; +https://eltociear-clean-read.hf.space)"})
            resp.raise_for_status()
    except httpx.HTTPStatusError as e:
        raise HTTPException(502, f"upstream returned {e.response.status_code}")
    except Exception as e:
        raise HTTPException(502, f"fetch failed: {type(e).__name__}: {e}")

    html = resp.text
    if len(html) > req.max_size:
        html = html[: req.max_size]

    import trafilatura
    markdown = trafilatura.extract(
        html,
        output_format="markdown",
        include_links=bool(req.include_links),
        include_tables=True,
        favor_recall=True,
    )
    if not markdown:
        raise HTTPException(422, "could not extract main content from this page")

    title = None
    try:
        meta = trafilatura.extract_metadata(html)
        if meta:
            title = meta.title
    except Exception:
        pass

    return {
        "url": str(resp.url),
        "title": title,
        "markdown": markdown,
        "word_count": len(markdown.split()),
        "fetched_at": datetime.utcnow().isoformat() + "Z",
    }


if __name__ == "__main__":
    import uvicorn
    port = int(os.environ.get("PORT", 8403))
    print(f"\n  clean-read API (x402 v2) starting on :{port}")
    print(f"  x402: {'ENABLED' if _x402_available else 'DISABLED (pip install x402[fastapi,evm,extensions])'}")
    print(f"  Facilitator: {FACILITATOR_URL}")
    print(f"  Wallet: {WALLET}\n")
    uvicorn.run(app, host="0.0.0.0", port=port)