SDKs
Elixir SDK
The official Vermin client for Elixir. This page is the SDK's README, rendered.
Official Elixir client for Vermin, the web-data API for AI: scrape, crawl, map, search, extract, brand. Built on Req.
def deps do
[{:vermin, "~> 0.1"}]
endQuick start#
client = Vermin.Client.new(api_key: "vmn_live_...") # or Vermin.Client.new() with VERMIN_API_KEY set
{:ok, %Vermin.ScrapeResult{data: doc, credits_used: credits}} =
Vermin.scrape(client, url: "https://example.com", formats: [:markdown, :metadata])
doc.markdown #=> "# Example Domain ..."
doc.metadata.title #=> "Example Domain"
credits #=> 1Every function returns {:ok, result} | {:error, %Vermin.Error{}} and has a bang variant (Vermin.scrape!/2, …) that returns the result or raises.
Params are keyword lists (or maps) with snake_case keys, sent as the API's camelCase (only_main_content: false → "onlyMainContent"); atom values are camelCased too (formats: [:raw_html] → "rawHtml"). :schema, :headers and webhook :metadata are sent verbatim. Responses are typed structs (Vermin.Document, Vermin.CrawlStatus, Vermin.Brand, …) with known enum values as atoms (tier: :render, status: :completed).
Configuration#
Vermin.Client.new(
api_key: "vmn_live_...", # default: $VERMIN_API_KEY
base_url: "https://api.vermin.dev", # default: $VERMIN_API_URL, then https://api.vermin.dev
receive_timeout: 60_000, # per request, default 90_000
max_retries: 3, # default 2
retry_base_ms: 500, retry_max_ms: 8_000,
req_options: [connect_options: [timeout: 5_000]] # merged into the Req request
)Endpoints#
# Scrape (a bare URL works too)
Vermin.scrape(client, "https://example.com")
Vermin.scrape(client,
url: "https://example.com",
formats: [:markdown, :chunks],
tier: :render,
actions: [[type: :click, selector: "#accept"], [type: :wait, ms: 500]]
)
# Map, optionally filtered by path globs
{:ok, %Vermin.MapResult{links: links}} = Vermin.map(client, url: "https://example.com", search: "pricing")
Vermin.map(client, url: "https://example.com", include_paths: ["/blog/*"], exclude_paths: ["/blog/tag/*"])
# Search, optionally scraping each hit into `document` (hits may carry `date` and `source`)
Vermin.search!(client, query: "elixir web scraping", limit: 5, scrape_options: [formats: [:markdown]])
# Extract structured data — `data` keeps your schema's keys
{:ok, %Vermin.ExtractResult{data: %{"plans" => plans}}} =
Vermin.extract(client,
urls: ["https://example.com/pricing"],
schema: %{type: "object", properties: %{plans: %{type: "array", items: %{type: "string"}}}},
prompt: "List plan names"
)
# Brand: `logo`/`icon` plus every candidate in `logos` (each with `type`, `background`,
# `source_url`); colours carry `text_color` and its WCAG `contrast`
{:ok, %Vermin.BrandResult{data: brand, cached: cached, stale: stale}} = Vermin.brand(client, "stripe.com")
Enum.map(brand.colors, & &1.hex)
Enum.filter(brand.logos || [], &(&1.background == :dark))
Vermin.brand(client, domain: "stripe.com", max_age: 0) # 0 forces a fresh extractionCrawling#
{:ok, job} = Vermin.Crawl.start(client, url: "https://docs.example.com", limit: 200, max_depth: 3)
# Live server-sent events as a lazy Stream of %Vermin.CrawlEvent{}
client
|> Vermin.Crawl.stream(job.id)
|> Enum.each(fn
%{type: :page, data: doc} -> IO.puts("scraped #{doc.url}")
%{type: :status, data: s} -> IO.puts("#{s.completed}/#{s.total}")
%{type: :done, data: s} -> IO.puts("done, #{s.credits_used} credits")
_other -> :ok
end)
# Lazy Stream of every document via polling + cursor pagination
client |> Vermin.Crawl.documents(job.id, poll_interval: 2_000) |> Stream.map(& &1.url) |> Enum.take(10)
# Block until finished and collect every page. A failed crawl is returned as
# {:ok, %Vermin.CrawlStatus{status: :failed, error: %Vermin.CrawlError{code: _, message: _}}}
{:ok, %Vermin.CrawlStatus{data: docs}} = Vermin.Crawl.wait(client, job.id, timeout: :timer.minutes(10))
# One page / cancel / start + wait in one call
Vermin.Crawl.get(client, job.id, cursor: nil, limit: 100)
Vermin.Crawl.cancel(client, job.id)
Vermin.crawl(client, [url: "https://example.com", limit: 10], poll_interval: 1_000)Vermin.Crawl.stream/3 connects when first enumerated, must be consumed by the enumerating process, and closes after the :done event (or when you halt, e.g. Enum.take/2). If the connection drops before :done it reconnects with Last-Event-ID (waiting the server's retry: delay, else backoff), up to :max_reconnects consecutive times (default :max_retries; the budget resets after each page/status event), then raises %Vermin.Error{code: :connection_error}. Resume yourself with Vermin.Crawl.stream(client, id, last_event_id: "41").
Errors#
case Vermin.scrape(client, url: "https://example.com") do
{:ok, result} -> result.credits_used
{:error, %Vermin.Error{code: :insufficient_credits}} -> :top_up
{:error, %Vermin.Error{code: code, status: status, request_id: id, message: msg}} ->
Logger.error("vermin #{code} (#{status}) #{msg} request_id=#{id}")
endVermin.Error carries :code (atom for known API codes - see Vermin.Error.codes/0, e.g. :capacity_exceeded, :not_configured, :extract_failed - :connection_error / :timeout for transport failures, :wait_timeout from Crawl.wait/3), :status, :request_id, :retry_after_ms, :message and :credits_used (always 0). Vermin.Error.retryable?/1 says whether retrying could help.
Webhooks#
Crawl webhooks carry a Vermin-Signature: t=<unix seconds>,v1=<hex> header, where v1 is the HMAC-SHA256 of "<t>.<raw body>" keyed with your signing secret (Dashboard → Webhooks). Verify it against the raw request body before trusting the payload. The helper compares in constant time, accepts any of several v1= values (secret rotation) and rejects timestamps more than 300 s from now (tolerance; 0 disables the check).
# raw_body: the request body exactly as received (keep it with a Plug.Parsers :body_reader)
case Vermin.Webhook.verify(raw_body, Plug.Conn.get_req_header(conn, "vermin-signature"), secret) do
{:ok, %{"type" => "crawl.page", "data" => data}} -> IO.puts(data["url"])
{:ok, _event} -> :ok
{:error, %Vermin.Error{code: :invalid_signature}} -> send_resp(conn, 400, "")
endVermin.Webhook.valid?/4 returns a boolean, verify!/4 raises, and sign/3 builds a header for your tests.
Retries and idempotency#
- 408, 429, 5xx (including
:capacity_exceeded) and transport errors are retried up to:max_retriestimes with jittered exponential backoff (base * 2^n, capped, jittered in[d/2, d]). Codes that would fail again are never retried, whatever the status::invalid_request,:unauthorized,:forbidden,:not_found,:not_implemented,:insufficient_credits,:spend_cap_reached,:blocked_url,:not_configured,:extract_failed. - A server
retry_after_ms(orRetry-Afterheader, seconds or HTTP date) is honoured as a floor, plus up to 10% jitter. - Every POST gets an auto-generated UUIDv4
Idempotency-Key, reused across retries of that call. Passidempotency_key: "..."in the params to set your own.
Testing your code#
Pass a Req.Test plug through :req_options:
client = Vermin.Client.new(api_key: "test", req_options: [plug: {Req.Test, MyApp.Vermin}])
Req.Test.stub(MyApp.Vermin, &Req.Test.json(&1, %{"success" => true, "data" => %{"url" => "u"}, "credits_used" => 1}))