openapi: 3.1.0
info:
  title: Wisepage API
  version: 0.1.0
  description: |
    Turn web pages into LLM-ready data for AI agents.

    Authenticate with `Authorization: Bearer <api key>`. Every response includes
    `credits_used`. Failed requests are not billed.

    Fetched pages are cached for up to an hour and shared across requests; use `max_age`
    to ask for a fresher copy.
servers:
  - url: https://api.wisepage.dev
    description: Production
  - url: http://localhost:3000
    description: Local development
security:
  - bearerAuth: []
paths:
  /v1/scrape:
    post:
      summary: Fetch a page or PDF as markdown, HTML or links (1 credit; PDFs 1 per 10 pages)
      requestBody:
        required: true
        content:
          application/json:
            schema: { $ref: "#/components/schemas/ScrapeRequest" }
            example: { url: "https://example.com", formats: ["markdown", "links"] }
      responses:
        "200":
          description: Page content
          content:
            application/json:
              schema: { $ref: "#/components/schemas/ScrapeResponse" }
        default: { $ref: "#/components/responses/Error" }
  /v1/search:
    post:
      summary: Search the web (Google results), optionally scraping each result (3 credits + 1 per scraped page)
      requestBody:
        required: true
        content:
          application/json:
            schema: { $ref: "#/components/schemas/SearchRequest" }
            example: { query: "best laptop 2026", limit: 5, scrape_results: true }
      responses:
        "200":
          description: Search results
          content:
            application/json:
              schema: { $ref: "#/components/schemas/SearchResponse" }
        default: { $ref: "#/components/responses/Error" }
  /v1/extract:
    post:
      summary: Extract structured JSON from a page using a JSON Schema (5 credits + tokens)
      description: |
        Costs 5 credits plus 4 per 1,000 input tokens and 20 per 1,000 output tokens, as
        reported in `usage` (typically 15-30 credits). 5 credits are needed to start; once the
        page is fetched, the most the call can cost for that page is reserved and the rest
        refunded (at most 267 credits, for the largest page and schema). When the balance cannot
        cover a 4,096-token answer, a 2,048- or 1,024-token answer is allowed instead; an answer
        that does not fit fails with 502 extract_truncated and is not billed. Pages longer than
        60,000 characters are cut, and `source.truncated` is true.
      requestBody:
        required: true
        content:
          application/json:
            schema: { $ref: "#/components/schemas/ExtractRequest" }
            example:
              url: "https://example.com/product/123"
              schema:
                type: object
                properties:
                  name: { type: string }
                  price: { type: number }
                  in_stock: { type: boolean }
                required: [name, price, in_stock]
      responses:
        "200":
          description: Extracted data
          content:
            application/json:
              schema: { $ref: "#/components/schemas/ExtractResponse" }
        default: { $ref: "#/components/responses/Error" }
  /v1/batch/scrape:
    post:
      summary: Fetch up to 50 pages in one call (1 credit per page fetched)
      description: |
        Scrapes every URL with the same options, several at a time, and returns the results
        in the order given. One credit per URL is reserved and the unused part refunded:
        URLs that fail get `success: false` and an error code, and are not billed. The batch
        stops starting new pages after about 60 seconds; those URLs report `time_budget_exceeded`.
      requestBody:
        required: true
        content:
          application/json:
            schema: { $ref: "#/components/schemas/BatchScrapeRequest" }
            example: { urls: ["https://example.com/a", "https://example.com/b"], formats: ["markdown"] }
      responses:
        "200":
          description: One result per URL
          content:
            application/json:
              schema: { $ref: "#/components/schemas/BatchScrapeResponse" }
        default: { $ref: "#/components/responses/Error" }
  /v1/map:
    post:
      summary: List a site's URLs from its sitemap and the links on a page (1 credit)
      description: |
        Returns URLs on the same site as `url` without fetching every page. Combines the
        sitemap (found via robots.txt or /sitemap.xml) with the links on the given page.
      requestBody:
        required: true
        content:
          application/json:
            schema: { $ref: "#/components/schemas/MapRequest" }
            example: { url: "https://docs.example.com", search: "api", limit: 100 }
      responses:
        "200":
          description: URLs found on the site
          content:
            application/json:
              schema: { $ref: "#/components/schemas/MapResponse" }
        default: { $ref: "#/components/responses/Error" }
  /v1/crawl:
    post:
      summary: Crawl a site and return each page as markdown (1 credit per page crawled)
      description: |
        Follows same-site links breadth first from `url`, up to `limit` pages and `max_depth`
        hops. Up to `limit` credits are reserved and the unused part is refunded: pages that
        fail are listed in `failed` and are not billed. A crawl stops after about 60 seconds;
        `stats.stopped_reason` says why it ended.
      requestBody:
        required: true
        content:
          application/json:
            schema: { $ref: "#/components/schemas/CrawlRequest" }
            example: { url: "https://docs.example.com", limit: 20, max_depth: 2, include_paths: ["/docs/*"] }
      responses:
        "200":
          description: Crawled pages
          content:
            application/json:
              schema: { $ref: "#/components/schemas/CrawlResponse" }
        default: { $ref: "#/components/responses/Error" }
  /v1/usage:
    get:
      summary: Remaining credits and recent usage (free)
      parameters:
        - { name: limit, in: query, schema: { type: integer, minimum: 1, maximum: 200, default: 50 } }
      responses:
        "200":
          description: Usage
          content:
            application/json:
              schema:
                type: object
                properties:
                  success: { type: boolean }
                  data:
                    type: object
                    properties:
                      credits_remaining: { type: integer }
                      recent:
                        type: array
                        items:
                          type: object
                          properties:
                            operation: { type: string }
                            credits: { type: integer }
                            success: { type: boolean }
                            duration_ms: { type: integer }
                            created_at: { type: string, format: date-time }
        default: { $ref: "#/components/responses/Error" }
  /v1/billing:
    get:
      summary: Current plan, credits and refill date (free)
      responses:
        "200":
          description: Billing state
          content:
            application/json:
              schema:
                type: object
                properties:
                  success: { type: boolean }
                  data:
                    type: object
                    properties:
                      plan: { type: string, enum: [free, hobby, standard] }
                      subscription_status: { type: [string, "null"], description: "Paddle status: active, past_due, paused, canceled" }
                      credits_remaining: { type: integer }
                      monthly_credits: { type: integer, description: Credits the balance is topped up to each month }
                      next_refill_at: { type: [string, "null"], format: date-time, description: Free plan only }
                      current_period_ends_at: { type: [string, "null"], format: date-time }
                      plans: { type: object, description: Plan catalog with monthly_credits and price_usd }
        default: { $ref: "#/components/responses/Error" }
  /v1/billing/checkout:
    post:
      summary: Start a subscription; returns a Paddle checkout link
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required: [plan]
              properties:
                plan: { type: string, enum: [hobby, standard] }
      responses:
        "200":
          description: Checkout link
          content:
            application/json:
              schema:
                type: object
                properties:
                  success: { type: boolean }
                  data: { type: object, properties: { checkout_url: { type: string } } }
        "409": { description: Already subscribed; use the billing portal to change plans }
        default: { $ref: "#/components/responses/Error" }
  /v1/billing/portal:
    post:
      summary: Link to the Paddle customer portal (update card, change or cancel)
      responses:
        "200":
          description: Portal link
          content:
            application/json:
              schema:
                type: object
                properties:
                  success: { type: boolean }
                  data: { type: object, properties: { portal_url: { type: string } } }
        "404": { description: No paid subscription on this account }
        default: { $ref: "#/components/responses/Error" }
  /webhooks/paddle:
    post:
      summary: Paddle notification receiver (verified with the Paddle-Signature header)
      security: []
      responses:
        "200": { description: "Processed: status is applied, duplicate or ignored" }
        "401": { description: Invalid signature }
  /mcp:
    post:
      summary: Remote MCP server (Streamable HTTP) exposing the scrape, batch_scrape, search, extract, map and crawl tools
      responses:
        "200": { description: JSON-RPC response }
components:
  securitySchemes:
    bearerAuth: { type: http, scheme: bearer }
  responses:
    Error:
      description: |
        Error. Common codes: 400 invalid_request / url_not_allowed / invalid_schema,
        401 missing_api_key / invalid_api_key, 402 insufficient_credits,
        403 blocked_by_robots / domain_blocked, 413 response_too_large / content_too_large,
        415 unsupported_content_type, 422 pdf_no_text / pdf_encrypted / pdf_invalid,
        429 rate_limited / domain_rate_limited, 502 fetch_failed,
        503 search_unavailable / extract_unavailable / browser_busy / pdf_busy, 504 fetch_timeout.
      content:
        application/json:
          schema:
            type: object
            properties:
              error:
                type: object
                properties:
                  code: { type: string }
                  message: { type: string }
                  details: {}
  schemas:
    ScrapeRequest:
      type: object
      required: [url]
      properties:
        url: { type: string, format: uri }
        formats:
          type: array
          items: { type: string, enum: [markdown, html, links] }
          default: [markdown]
        only_main_content: { type: boolean, default: true, description: Strip navigation, headers, footers and ads }
        render:
          type: string
          enum: [auto, always, never]
          default: auto
          description: Use a headless browser. `auto` renders only when the page needs JavaScript.
        wait_for_ms: { type: integer, minimum: 0, maximum: 10000 }
        timeout_ms: { type: integer, minimum: 1000, maximum: 60000 }
        respect_robots: { type: boolean, default: false }
        max_age: { $ref: "#/components/schemas/MaxAge" }
    Metadata:
      type: object
      properties:
        title: { type: [string, "null"] }
        description: { type: [string, "null"] }
        language: { type: [string, "null"] }
        site_name: { type: [string, "null"] }
        author: { type: [string, "null"] }
        published_time: { type: [string, "null"] }
        canonical_url: { type: [string, "null"] }
        image: { type: [string, "null"] }
        source_url: { type: string }
        content_type: { type: string }
    ScrapeResponse:
      type: object
      properties:
        success: { type: boolean }
        credits_used: { type: integer }
        data: { $ref: "#/components/schemas/ScrapeResult" }
    ScrapeResult:
      type: object
      properties:
        url: { type: string, description: Final URL after redirects }
        status_code: { type: integer }
        rendered: { type: boolean, description: Whether a headless browser was used }
        markdown: { type: string }
        html: { type: string }
        links: { type: array, items: { type: string } }
        metadata: { $ref: "#/components/schemas/Metadata" }
        cached: { type: boolean, description: Whether this copy came from the page cache }
        fetched_at: { type: string, format: date-time, description: When the page was actually fetched }
        pdf: { $ref: "#/components/schemas/PdfInfo" }
    PdfInfo:
      type: object
      description: |
        Present when the URL was a PDF. Its text is in `markdown`, rebuilt from the layout
        (headings, paragraphs, lists and tables; running headers, footers and page numbers
        dropped), with a `<!-- page N -->` marker between pages; `html` is not returned. Up to 100 pages and 20 MB are read, billed at
        1 credit per 10 pages read (at least 1).
      properties:
        pages: { type: integer, description: Pages in the document }
        pages_parsed: { type: integer, description: Pages read }
        truncated: { type: boolean, description: Whether some pages were not read }
    MaxAge:
      type: integer
      minimum: 0
      maximum: 86400
      default: 3600
      description: |
        Accept a cached copy of the page up to this many seconds old; 0 always fetches it fresh.
        Cached copies are checked against the blocklist and robots.txt like a fresh fetch, and
        are billed the same.
    SearchRequest:
      type: object
      required: [query]
      properties:
        query: { type: string, maxLength: 400 }
        limit: { type: integer, minimum: 1, maximum: 10, default: 5 }
        country: { type: string, description: Two-letter country code }
        language: { type: string }
        scrape_results: { type: boolean, default: false }
    SearchResponse:
      type: object
      properties:
        success: { type: boolean }
        credits_used: { type: integer }
        data:
          type: array
          items:
            type: object
            properties:
              title: { type: string }
              url: { type: string }
              description: { type: string }
              markdown: { type: string, description: Present when scrape_results is true and the page was fetched }
              scraped: { type: boolean }
              error: { type: string }
    ExtractRequest:
      type: object
      required: [url, schema]
      properties:
        url: { type: string, format: uri }
        schema: { type: object, description: 'JSON Schema for the output, under 20,000 characters. The root must be "type": "object".' }
        prompt: { type: string, maxLength: 4000 }
        render: { type: string, enum: [auto, always, never], default: auto }
        timeout_ms: { type: integer, minimum: 1000, maximum: 60000 }
        max_age: { $ref: "#/components/schemas/MaxAge" }
    ExtractResponse:
      type: object
      properties:
        success: { type: boolean }
        credits_used: { type: integer }
        data: { description: Object matching the requested schema }
        source:
          type: object
          properties:
            url: { type: string }
            title: { type: [string, "null"] }
            truncated: { type: boolean, description: The page was longer than 60,000 characters and only its start was read }
        usage:
          type: object
          properties:
            input_tokens: { type: integer }
            output_tokens: { type: integer }
    BatchScrapeRequest:
      type: object
      required: [urls]
      properties:
        urls: { type: array, minItems: 1, maxItems: 50, items: { type: string, format: uri } }
        formats:
          type: array
          items: { type: string, enum: [markdown, html, links] }
          default: [markdown]
        only_main_content: { type: boolean, default: true }
        render: { type: string, enum: [auto, always, never], default: auto }
        wait_for_ms: { type: integer, minimum: 0, maximum: 10000 }
        timeout_ms: { type: integer, minimum: 1000, maximum: 60000 }
        respect_robots: { type: boolean, default: false }
        max_age: { $ref: "#/components/schemas/MaxAge" }
    BatchScrapeResponse:
      type: object
      properties:
        success: { type: boolean }
        credits_used: { type: integer, description: "One per result with `success: true`" }
        data:
          type: object
          properties:
            results:
              type: array
              description: Same order as `urls`
              items:
                type: object
                properties:
                  url: { type: string, description: The URL as requested }
                  success: { type: boolean }
                  data: { $ref: "#/components/schemas/ScrapeResult", description: Present when success is true }
                  error: { type: string, description: Error code when success is false }
            stats:
              type: object
              properties:
                succeeded: { type: integer }
                failed: { type: integer }
                stopped_reason: { type: string, enum: [done, time_budget] }
    MapRequest:
      type: object
      required: [url]
      properties:
        url: { type: string, format: uri, description: Any page on the site to map }
        limit: { type: integer, minimum: 1, maximum: 5000, default: 500 }
        search: { type: string, maxLength: 200, description: Only return URLs containing this text }
        include_subdomains: { type: boolean, default: false }
        include_paths: { $ref: "#/components/schemas/PathPatterns" }
        exclude_paths: { $ref: "#/components/schemas/PathPatterns" }
        sitemap:
          type: string
          enum: [include, skip, only]
          default: include
          description: Use the sitemap as well as the page's links, ignore it, or use only the sitemap
        include_pdfs: { type: boolean, default: false, description: Also list links to PDF files }
    MapResponse:
      type: object
      properties:
        success: { type: boolean }
        credits_used: { type: integer }
        data:
          type: object
          properties:
            url: { type: string }
            links: { type: array, items: { type: string } }
    CrawlRequest:
      type: object
      required: [url]
      properties:
        url: { type: string, format: uri, description: The page to start crawling from }
        limit: { type: integer, minimum: 1, maximum: 50, default: 10, description: Maximum number of pages to fetch }
        max_depth: { type: integer, minimum: 0, maximum: 5, default: 2, description: How many links away from the start page to follow }
        include_subdomains: { type: boolean, default: false }
        include_paths: { $ref: "#/components/schemas/PathPatterns" }
        exclude_paths: { $ref: "#/components/schemas/PathPatterns" }
        use_sitemap: { type: boolean, default: false, description: Also queue pages listed in the site's sitemap }
        only_main_content: { type: boolean, default: true }
        render: { type: string, enum: [auto, always, never], default: auto }
        respect_robots: { type: boolean, default: true }
        max_age: { $ref: "#/components/schemas/MaxAge" }
        include_pdfs: { type: boolean, default: false, description: Also read linked PDF files (1 credit per 10 PDF pages) }
    CrawlResponse:
      type: object
      properties:
        success: { type: boolean }
        credits_used: { type: integer, description: One per page in `pages` (PDFs one per 10 pages read) }
        data:
          type: object
          properties:
            pages:
              type: array
              items:
                type: object
                properties:
                  url: { type: string }
                  markdown: { type: string }
                  metadata: { $ref: "#/components/schemas/Metadata" }
                  pdf: { $ref: "#/components/schemas/PdfInfo" }
            failed:
              type: array
              items:
                type: object
                properties:
                  url: { type: string }
                  error: { type: string, description: Error code, e.g. fetch_failed or blocked_by_robots }
            stats:
              type: object
              properties:
                pages_crawled: { type: integer }
                pages_failed: { type: integer }
                urls_discovered: { type: integer }
                stopped_reason: { type: string, enum: [limit, exhausted, time_budget] }
    PathPatterns:
      type: array
      maxItems: 20
      items: { type: string }
      description: 'Path patterns where `*` matches anything, e.g. "/docs/*"'
