One snippet in your middleware.
Every AI visit, captured.
Next.js runs your code on every request, so it's the perfect place to measure AI agents. Drop in our middleware (proxy.ts on Next 16) and each request fires a background event to Arrivl, no impact on your render.
A single middleware file and an environment variable. Your AI coding agent can wire it in from our prompt.
Install guide
Install Next.js, start to finish
Next.js middleware runs before every request reaches your app, which makes it the one pasted snippet that can do more than watch. The embedded install below reports your traffic AND applies the optimizations you approve in the dashboard. The original reporting-only middleware is still supported and still works — it is kept further down for anyone already running it.
Monitoring + automatic optimization — this install can apply the changes you approve.
01Install guide
Prerequisites
- 01An Arrivl website key
- 02Next.js 13+ (App Router or Pages Router). Next 16 renamed
middleware.tstoproxy.ts— both are covered below.
02Install guide
Install
Add the file
Save this as
arrivl-edge.jsnext to yourproxy.ts(Next 16) ormiddleware.ts(Next 15). It never needs editing — the key and every option come from your own code.javascript
// arrivl-nextjs-embed-rev: 1 // Arrivl edge integration for Next.js (proxy.ts on Next 16, middleware.ts on // Next 15). Save this file next to that file, then wrap your handler: // // import { withArrivl } from "./arrivl-edge.js"; // // export const proxy = withArrivl(async (request) => { // // your existing middleware, unchanged (or omit the argument entirely) // }); // // export const config = { matcher: [ ...paste the list from the guide... ] }; // // Then set ARRIVL_WEBSITE_KEY in your environment. // // WHAT IT DOES // 1. Reports AI-agent and human pageviews to Arrivl, fire-and-forget, from // inside your app. Server-side, so crawlers that never run JavaScript // are counted. // 2. Answers the AI auxiliary paths (/llms.txt, /robots.txt, /ai.txt, *.md // alternates, ...) from the optimization bundle you approved in the // Arrivl dashboard, and applies the response headers and redirects it // contains. // // WHAT IT WILL NOT DO // * It never blocks your response. Reporting rides on the fetch event's // waitUntil when your platform provides one. // * With no active bundle it returns exactly what your handler returned // (one identifying header is added on tracked paths; pass // { identify: false } to suppress even that). // * If anything here throws, your handler's response is returned as-is. // Analytics must never break the site. // * It cannot invent a redirect or shadow a page of yours on its own. // Payload-serving is restricted to the fixed list of AI auxiliary paths in // RESERVED_STATIC_PATHS below, and the only rule that can move one of your // existing paths (path_redirect) is an EXACT path -> path pair that you // see and approve in the dashboard before it goes live. Pass // { optimize: false } to run reporting only. // // Questions or an audit of this file: hello@arrivl.ai // The one import. Used only to build the "continue" response when a header // rule has something to add and your handler returned nothing; every other // response here is a plain web Response, which Next accepts from middleware. import { NextResponse } from "next/server"; /** Bump me only via Arrivl's release process — reported as sv= on every event * so Arrivl can tell which vintage of this file you are running. */ export var ARRIVL_SNIPPET_VERSION = 1; var SOURCE = "nextjs_embed"; var DEFAULT_API_HOST = "arrivl.ai"; var DEFAULT_KEY_VAR = "ARRIVL_WEBSITE_KEY"; // Bundle fetch containment. A slow Arrivl must never become a slow customer // site: one fetch, capped, and a failure is remembered so the next requests // skip it entirely instead of each paying the timeout. var BUNDLE_TIMEOUT_MS = 2500; var BUNDLE_TTL_MS = 60000; // Paths worth recording. Mirrors Arrivl's shared filter. This is a SECOND // layer: your exported matcher decides what reaches this file at all. var FORCE_INCLUDE = ["/.well-known/ai-plugin.json","/AGENTS.md","/ai.txt","/catalog.json","/llms-full.txt","/llms.txt","/robots.txt","/sitemap-md.xml","/sitemap.xml"]; var SKIP_EXT = /\.(png|jpe?g|svg|gif|webp|ico|css|js|mjs|map|woff2?|ttf|otf|eot|txt|xml|json|webmanifest)$/i; var SITEMAP_XML = /[/][^/]*sitemap[^/]*[.]xml$/i; function shouldTrack(pathname) { if (FORCE_INCLUDE.indexOf(pathname) !== -1) return true; if (pathname.indexOf("/.well-known/") === 0 || pathname.indexOf("/sitemap-") === 0) return true; if (SITEMAP_XML.test(pathname)) return true; if (pathname.indexOf("/api/") === 0 || pathname.indexOf("/_next/") === 0) return false; if (pathname === "/favicon.ico") return false; if (SKIP_EXT.test(pathname)) return false; return true; } // The ONLY paths a bundle may answer with its own payload. Your real pages and // routes are unreachable from here by construction. var RESERVED_STATIC_PATHS = ["/llms.txt","/llms-full.txt","/ai.txt","/robots.txt","/sitemap.xml","/sitemap-md.xml","/AGENTS.md","/.well-known/ai-plugin.json","/catalog.json"]; function staticServeAllowed(p) { if (RESERVED_STATIC_PATHS.indexOf(p) !== -1) return true; if (p.length > 3 && p.indexOf(".md", p.length - 3) !== -1) return true; return false; } // A path where the bundle may answer INSTEAD of your app must be resolved // before your handler runs; everywhere else the bundle load and your handler // run concurrently, so ordinary pages pay no added latency. function mustResolveFirst(pathname) { return pathname.indexOf("/go/") === 0 || staticServeAllowed(pathname); } // A malformed pattern from config makes its own rule inert — never the site. var RE_CACHE = {}; function testRe(src, s) { try { var re = RE_CACHE[src] || (RE_CACHE[src] = new RegExp(src, "i")); return re.test(s); } catch (e) { return false; } } function pathMatches(match, pathname) { if (!match) return false; if (match.path_exact !== undefined) return match.path_exact === pathname; if (Array.isArray(match.path_alternatives)) return match.path_alternatives.indexOf(pathname) !== -1; if (Array.isArray(match.path_enumeration)) return match.path_enumeration.indexOf(pathname) !== -1; if (typeof match.path_regex === "string") return testRe(match.path_regex, pathname); if (Array.isArray(match.path_regex_alternatives)) { for (var ri = 0; ri < match.path_regex_alternatives.length; ri++) { if (testRe(match.path_regex_alternatives[ri], pathname)) return true; } } return false; } function staticServeBody(rule, pathname) { if (rule.body !== undefined && rule.body !== null) return rule.body; if (rule.payloadMap) { if (Object.prototype.hasOwnProperty.call(rule.payloadMap, pathname)) return rule.payloadMap[pathname]; try { var dec = decodeURIComponent(pathname); if (Object.prototype.hasOwnProperty.call(rule.payloadMap, dec)) return rule.payloadMap[dec]; } catch (e) {} } return null; } // Externally-visible URL. `request.url` carries the raw Host header, and behind // a reverse proxy — Vercel, Fly, Railway, Render, an ALB, nginx, any container // platform — that Host is the INTERNAL address, frequently localhost. Arrivl // stores that hostname on every event and only marks the install verified once // it matches your real domain, so reporting the internal one means the site // never verifies. Proxies APPEND, so the leftmost value is the public one. function publicUrl(request) { try { var fwdHost = (request.headers.get("x-forwarded-host") || "").split(",")[0].trim(); if (!fwdHost) return request.url; var fwdProto = (request.headers.get("x-forwarded-proto") || "").split(",")[0].trim() || "https"; return request.url.replace(/^https?:\/\/[^/]+/, fwdProto + "://" + fwdHost); } catch (e) { return request.url; } } function clientIp(request) { var ip = request.headers.get("cf-connecting-ip"); if (ip) return ip; var fwd = request.headers.get("x-forwarded-for"); if (fwd) { var first = fwd.split(",")[0]; if (first) return first.trim(); } return request.headers.get("x-real-ip") || ""; } // Fire-and-forget pageview. Returns a promise the caller hands to waitUntil; // it never rejects. function fireIntake(cfg, request, res, tf) { var ip = clientIp(request); // Arrivl requires a non-empty ip; with none resolved the event would be // rejected anyway, so skip the call rather than spend it. if (!ip) return Promise.resolve(); var base = { url: publicUrl(request), userAgent: request.headers.get("user-agent") || "", ref: request.headers.get("referer") || "", ip: ip, websiteKey: cfg.websiteKey, source: SOURCE, sv: String(ARRIVL_SNIPPET_VERSION), }; var acc = request.headers.get("accept"); if (acc) base.accept = acc; // Web Bot Auth signature headers — forwarded raw when present. var sg = request.headers.get("signature"); if (sg) base.sig = sg; var sgi = request.headers.get("signature-input"); if (sgi) base.sigInput = sgi; var sga = request.headers.get("signature-agent"); if (sga) base.sigAgent = sga; // Next strips the internal Flight headers (rsc, next-router-prefetch) from // the request object inside middleware, so those are not readable here. The // Speculation-Rules / link-prefetch hints are, and the `_rsc` query marker // rides along in the url, which is where Arrivl reads it from. var purpose = request.headers.get("sec-purpose") || request.headers.get("purpose") || ""; if (purpose.indexOf("prefetch") !== -1) base.pf = "1"; if (request.method === "HEAD") base.mt = "HEAD"; // Status is reported ONLY when this file produced the response. A pass- // through has no status to honestly claim: your app has not answered yet. if (res && typeof res.status === "number") { base.status = String(res.status); try { var len = res.headers.get("content-length"); if (len) base.bytes = len; } catch (e) {} } var params = new URLSearchParams(base); try { if (tf && tf.a && tf.a.length) params.set("transform_attempted", tf.a.slice(0, 20).join(",")); if (tf && tf.p && tf.p.length) params.set("transform_applied", tf.p.slice(0, 20).join(",")); } catch (e) {} try { return fetch(cfg.intakeUrl + "?" + params.toString()).then(function () {}, function () {}); } catch (e) { return Promise.resolve(); } } // One bundle per app, cached in module scope for BUNDLE_TTL_MS. There is no // Cache API here (Next middleware runs on the Node runtime in Next 16 and on // the Edge runtime in Next 15; neither exposes Cloudflare's caches.default), // so this in-process memo IS the cache — including for failures, so a broken // Arrivl costs at most one capped fetch per instance per minute. var MEMO = { at: 0, value: null }; var INFLIGHT = null; function loadBundle(cfg) { var now = Date.now(); if (MEMO.value && now - MEMO.at < BUNDLE_TTL_MS) return Promise.resolve(MEMO.value); // Collapse a burst of concurrent requests onto ONE fetch. if (INFLIGHT) return INFLIGHT; function remember(value) { MEMO = { at: Date.now(), value: value }; INFLIGHT = null; return value; } var signal; try { signal = AbortSignal.timeout(BUNDLE_TIMEOUT_MS); } catch (e) { signal = undefined; } INFLIGHT = fetch(cfg.bundleUrl, signal ? { signal: signal } : {}).then( function (res) { if (!res || !res.ok) return remember({ rules: [], version: 0, state: "err:status" }); return res.json().then( function (body) { if (body && Array.isArray(body.rules)) { return remember({ rules: body.rules, version: body.version || 0, state: "ok" }); } return remember({ rules: [], version: 0, state: "shape" }); }, function () { return remember({ rules: [], version: 0, state: "err:parse" }); }, ); }, function (e) { return remember({ rules: [], version: 0, state: "err:" + ((e && e.name) || "fetch") }); }, ); return INFLIGHT; } /** event.waitUntil when the platform gave us one; a detached promise * otherwise. Never throws, so a host without an event cannot break the * request. */ function waitUntil(event, promise) { try { if (event && typeof event.waitUntil === "function") { event.waitUntil(promise); return; } } catch (e) {} try { Promise.resolve(promise).catch(function () {}); } catch (e) {} } /** Call the wrapped handler. Absent handler = "continue", which is what an * install with no middleware of its own looks like. */ function callHandler(handler, request, event) { if (typeof handler !== "function") return Promise.resolve(undefined); return Promise.resolve(handler(request, event)); } function resolveConfig(options) { var opts = options || {}; var key = opts.websiteKey; if (!key) { try { key = process.env[opts.websiteKeyVar || DEFAULT_KEY_VAR]; } catch (e) { key = null; } } if (!key || typeof key !== "string") return null; var host = opts.apiHost || DEFAULT_API_HOST; return { websiteKey: key, intakeUrl: "https://" + host + "/api/v1/intake/pageview", bundleUrl: "https://" + host + "/api/v1/internal/transform-bundle?key=" + key, assetUrl: "https://" + host + "/api/v1/internal/transform-asset?key=" + key, optimize: opts.optimize !== false, identify: opts.identify !== false, }; } /** * Wrap your Next.js proxy/middleware. * * @param handler your existing middleware function, or nothing at all. * @param options { websiteKey?, websiteKeyVar?, apiHost?, optimize?, identify? } * @returns the function to export as `proxy` (Next 16) or * `middleware` (Next 15). */ export function withArrivl(handler, options) { return async function arrivlProxy(request, event) { var cfg = resolveConfig(options); // No key configured: behave as if this wrapper were not here at all. if (!cfg) return callHandler(handler, request, event); var pathname; try { pathname = new URL(request.url).pathname; } catch (e) { return callHandler(handler, request, event); } var tracked = shouldTrack(pathname); var method = request.method; var getLike = method === "GET" || method === "HEAD"; var isHead = method === "HEAD"; var consult = cfg.optimize && getLike && tracked; var rules = []; var bundleVersion = 0; var bundleState = "skip"; var resPromise = null; try { if (consult && mustResolveFirst(pathname)) { // The bundle may answer this path itself, so resolve it BEFORE running // your handler — your handler is not called at all when a reserved AI // path is served from the bundle. var b1 = await loadBundle(cfg); rules = b1.rules; bundleVersion = b1.version; bundleState = b1.state + ":" + rules.length; } else if (consult) { // Everywhere else: your handler and the bundle load run at the same // time, so the bundle costs no added latency. resPromise = callHandler(handler, request, event); resPromise.catch(function () {}); var b2 = await loadBundle(cfg); rules = b2.rules; bundleVersion = b2.version; bundleState = b2.state + ":" + rules.length; } } catch (e) { rules = []; bundleState = "err:wrapper"; } var tfAttempted = []; try { // ---- go_redirect: measurable "next step" links -------------------- if (pathname.indexOf("/go/") === 0 && getLike) { for (var g = 0; g < rules.length; g++) { var gr = rules[g]; if (gr.type !== "go_redirect" || gr.enabled === false || !gr.targets) continue; var gkey = pathname.slice(4).replace(/\/$/, "").toLowerCase(); var dest = Object.prototype.hasOwnProperty.call(gr.targets, gkey) ? gr.targets[gkey] : null; if (typeof dest !== "string" || dest.charAt(0) !== "/" || dest.indexOf("//") === 0) continue; var goRes = redirectTo(request, dest, 302, "go_redirect"); waitUntil(event, fireIntake(cfg, request, goRes, { a: [String(gr.id || "go_redirect")], p: [String(gr.id || "go_redirect")] })); return goRes; } } // ---- path_redirect: one exact alias -> its canonical path ---------- // EXACT pathname only. There is no pattern form, so a path you did not // review can never be redirected, and Arrivl's publish validation // forbids chains (a destination may not be another rule's source). for (var pr = 0; pr < rules.length; pr++) { var prr = rules[pr]; if (prr.type !== "path_redirect" || prr.enabled === false) continue; if (typeof prr.from !== "string" || typeof prr.to !== "string") continue; if (prr.from !== pathname || prr.to === prr.from) continue; if (prr.to.charAt(0) !== "/" || prr.to.indexOf("//") === 0) continue; var prRes = redirectTo(request, prr.to, prr.status === 301 ? 301 : 301, "path_redirect"); if (tracked) waitUntil(event, fireIntake(cfg, request, prRes, { a: [String(prr.id || "path_redirect")], p: [String(prr.id || "path_redirect")] })); return prRes; } // ---- static_serve: answer a reserved AI path ----------------------- for (var i = 0; i < rules.length; i++) { var rule = rules[i]; if (rule.type !== "static_serve" || rule.enabled === false) continue; if (!staticServeAllowed(pathname)) break; if (!pathMatches(rule.match, pathname)) continue; var body = staticServeBody(rule, pathname); if (body === null) { if (!isHead) tfAttempted.push(String(rule.id || "static_serve")); continue; } var ssRes = textResponse(body, rule.contentType || "text/plain; charset=utf-8", "static_serve", isHead); if (tracked) waitUntil(event, fireIntake(cfg, request, ssRes, isHead ? null : { a: tfAttempted.concat([String(rule.id || "static_serve")]), p: [String(rule.id || "static_serve")] })); return ssRes; } // ---- static_serve_ref: same, body fetched per-asset ----------------- for (var sr = 0; sr < rules.length; sr++) { var srr = rules[sr]; if (srr.type !== "static_serve_ref" || srr.enabled === false) continue; if (!staticServeAllowed(pathname)) break; if (!pathMatches(srr.match, pathname)) continue; var refBody = null; try { var assetUrl = cfg.assetUrl + "&path=" + encodeURIComponent(pathname) + "&v=" + bundleVersion; var aSignal; try { aSignal = AbortSignal.timeout(BUNDLE_TIMEOUT_MS); } catch (e) { aSignal = undefined; } var fetched = await fetch(assetUrl, aSignal ? { signal: aSignal } : {}); if (fetched && fetched.status === 200) refBody = await fetched.text(); } catch (e) { refBody = null; } // Fail-open: a missing asset falls through to your app, never a 500. if (refBody === null) { if (!isHead) tfAttempted.push(String(srr.id || "static_serve_ref")); continue; } var refRes = textResponse(refBody, srr.contentType || "text/markdown; charset=utf-8", "static_serve_ref", isHead); if (tracked) waitUntil(event, fireIntake(cfg, request, refRes, isHead ? null : { a: tfAttempted.concat([String(srr.id || "static_serve_ref")]), p: [String(srr.id || "static_serve_ref")] })); return refRes; } // ---- bot_status: answer a named crawler with a fixed status --------- var uaTokens = (request.headers.get("user-agent") || "").toLowerCase().split(/[^a-z0-9-]+/); var uaSet = {}; for (var ti = 0; ti < uaTokens.length; ti++) if (uaTokens[ti]) uaSet[uaTokens[ti]] = 1; for (var bs = 0; bs < rules.length; bs++) { var bsr = rules[bs]; if (bsr.type !== "bot_status" || bsr.enabled === false) continue; if (!Array.isArray(bsr.botNames) || !Array.isArray(bsr.pathPrefixes) || !bsr.pathPrefixes.length) continue; var uaHit = false; for (var ui = 0; ui < bsr.botNames.length; ui++) { if (uaSet[String(bsr.botNames[ui]).toLowerCase()] === 1) { uaHit = true; break; } } if (!uaHit) continue; var pathHit = false; for (var pi = 0; pi < bsr.pathPrefixes.length; pi++) { if (pathname.indexOf(bsr.pathPrefixes[pi]) === 0) { pathHit = true; break; } } if (!pathHit) continue; var st = (bsr.status === 403 || bsr.status === 404 || bsr.status === 410 || bsr.status === 429) ? bsr.status : 410; var bsRes = new Response(isHead ? null : String(st) + "\n", { status: st, headers: { "content-type": "text/plain; charset=utf-8", "x-arrivl-transform": "bot_status" }, }); if (tracked) waitUntil(event, fireIntake(cfg, request, bsRes, { a: tfAttempted.concat([String(bsr.id || "bot_status")]), p: [String(bsr.id || "bot_status")] })); return bsRes; } } catch (e) { // Any failure above falls through to your handler untouched. tfAttempted = []; } // ---- your handler runs here (exactly once, always) ------------------- var response = resPromise ? await resPromise : await callHandler(handler, request, event); // ---- response_header_set — the one rule type that reaches a response // this file never sees the body of. Next forwards headers set on the // "continue" response to the client; that is the documented contract. var headerRules = []; try { for (var j = 0; j < rules.length; j++) { var r = rules[j]; if (r.enabled === false || r.type !== "response_header_set") continue; if (!pathMatches(r.match, pathname)) continue; if (typeof r.headerName !== "string" || typeof r.headerValue !== "string") continue; headerRules.push(r); tfAttempted.push(String(r.id || r.type)); } } catch (e) { headerRules = []; } var needsResponse = tracked && (cfg.identify || headerRules.length > 0); if (!response && needsResponse) { // Your handler said "continue" and we have something to add. An explicit // NextResponse.next() is exactly what returning nothing means, so this // changes the request's routing not at all — it only gives the headers // somewhere to live. try { response = NextResponse.next(); } catch (e) { response = undefined; } } var tfApplied = []; if (response && headerRules.length) { try { for (var k = 0; k < headerRules.length; k++) { var hr = headerRules[k]; if (hr.mode === "append") response.headers.append(hr.headerName, hr.headerValue); else response.headers.set(hr.headerName, hr.headerValue); tfApplied.push(String(hr.id || "response_header_set")); } } catch (e) { // Fail-open: a header must never break your page. } } if (tracked) waitUntil(event, fireIntake(cfg, request, null, tfAttempted.length ? { a: tfAttempted, p: tfApplied } : null)); if (response && tracked && cfg.identify) { try { response.headers.set("x-arrivl-embed", "sv=" + ARRIVL_SNIPPET_VERSION + ";bundle=" + bundleState); } catch (e) {} } return response; }; } /** A redirect Response. Built by hand rather than via NextResponse so this * file needs no import on the hot path; Next accepts any Response. */ function redirectTo(request, to, status, kind) { var location = to; try { var u = new URL(request.url); // Carry the query string over: an alias redirect must not drop the // campaign parameters the visitor arrived with. location = to + (u.search || ""); } catch (e) {} return new Response(null, { status: status, headers: { Location: location, "Cache-Control": "no-store", "x-arrivl-transform": kind }, }); } function textResponse(body, contentType, kind, isHead) { var text = String(body); var bytes = new TextEncoder().encode(text).length; return new Response(isHead ? null : text, { status: 200, headers: { "content-type": contentType, "x-arrivl-transform": kind, "content-length": String(bytes), }, }); }Wrap your middleware
Import it and wrap whatever you export today. If you have no middleware yet, call it with no arguments. Paste the matcher exactly as shown — Next.js reads it at build time by static analysis, so importing it from another file compiles cleanly and is then silently ignored.
proxy.ts
// proxy.ts (Next.js 16) — on Next.js 15 the file is middleware.ts and the // export is named `middleware` instead of `proxy`. Nothing else changes. import { withArrivl } from "./arrivl-edge.js"; export const proxy = withArrivl(async (request) => { // Your existing middleware goes here, completely unchanged. // Return nothing (or NextResponse.next()) to let the request continue. }); // Paste this matcher as-is. Next.js reads it at BUILD time by static analysis, // so it has to be a literal array here — importing it from another file // compiles fine and is silently ignored. export const config = { matcher: [ "/.well-known/ai-plugin.json", "/AGENTS.md", "/ai.txt", "/catalog.json", "/llms-full.txt", "/llms.txt", "/robots.txt", "/sitemap-md.xml", "/sitemap.xml", "/go/:path*", "/sitemap-:path*", "/.well-known/:path*", "/((?!api|_next/static|_next/image|favicon.ico|manifest.webmanifest|.*\\.(?:png|jpg|jpeg|svg|gif|webp|ico|css|js|mjs|map|woff2?|ttf|otf|eot|txt|xml|json)$).*)", ], }; // Then set your website key in the environment (never in source): // ARRIVL_WEBSITE_KEY=ak_YOUR_WEBSITE_KEYSet your website key
Add it to your environment (Vercel project settings, your
.env.local, your container config — wherever your other secrets live):.env.local
ARRIVL_WEBSITE_KEY=ak_YOUR_WEBSITE_KEY
Deploy, then load a page
That first real event is what flips your project to Connected — writing the code is not enough. To confirm the optimization half is live, request any page and look for the
x-arrivl-embedresponse header.bash
curl -sI https://yoursite.com/robots.txt | grep -i x-arrivl-embed # x-arrivl-embed: sv=1;bundle=ok:0 # sv = which version of arrivl-edge.js you are running # bundle = ok:<n>, the number of approved rules currently loaded # (ok:0 is correct until you approve your first optimization)
03Install guide
What it does to your responses
- 01With no approved bundle, nothing. Your response comes back with its body, status and headers exactly as your app produced them. One
x-arrivl-embedheader is added so a single request can verify the install; pass{ identify: false }to suppress even that. - 02It answers the AI auxiliary paths —
/llms.txt,/robots.txt,/ai.txt,/sitemap.xml, per-page.mdalternates — from the bundle you approved. It cannot answer any other path, so a real page of yours can never be shadowed. - 03It applies approved response headers and exact path redirects. A redirect rule is one explicit
/from→/topair that you see and approve before it goes live; there is no pattern form. - 04It never blocks your response. Reporting rides on
waitUntil, the bundle is fetched at most once a minute per instance with a 2.5s cap, and on any failure your handler's response is returned untouched.
04Install guide
What this host cannot do
- 01Next.js middleware decides whether a request continues — it never receives the HTML your app produces. So the optimizations that rewrite page content (JSON-LD injection,
<head>tags, JSON-LD edits, appended text, link rewriting, status-code overrides) are not available here; Arrivl knows that and simply never proposes them for a Next.js install. - 02If you want those too, the Cloudflare installs sit in front of the whole response and can rewrite it.
05Install guide
Verify
bash
curl "https://arrivl.ai/api/v1/intake/pageview\
?url=https://yoursite.com/test\
&userAgent=Mozilla/5.0%20(compatible;%20GPTBot/1.0)\
&ref=\
&accept=text/html\
&ip=1.2.3.4\
&websiteKey=ak_YOUR_WEBSITE_KEY"
# Should return: {"ok": true}01Nothing is being tracked
Make sure the file is at the project root (next to package.json), that ARRIVL_WEBSITE_KEY is set in the deployed environment rather than only locally, and that you pasted the config.matcher array literally.
02The x-arrivl-embed header says ok:0
That is the healthy state until you approve your first optimization: the module is running and holding zero rules. It changes to ok:<n> within a minute of approving a bundle.
03I am on Next.js 15, not 16
Name the file middleware.ts and export middleware instead of proxy. Nothing else changes — export const middleware = withArrivl(...).
04I already pasted the old reporting-only middleware
It keeps working exactly as before; there is no deadline and nothing breaks. Switching is worth it only if you want the optimization half — replace your middleware with the version above.
05Will this slow down my site?
For ordinary pages, no: your handler and the bundle load run at the same time, and reporting is fired in the background. Only the AI auxiliary paths wait for the bundle, because those are the paths it may answer itself.
07Install guide
Reporting-only middleware (the original install)
Still fully supported. Use this if you want traffic measurement and nothing else — it reports the same events, so your dashboard and weekly report are identical. It cannot serve optimizations, so no optimization proposals will be generated for a project installed this way.
08Install guide
middleware.ts — reporting only
middleware.ts
// middleware.ts (Next.js 15) or proxy.ts (Next.js 16)
// Next 16 note: rename the function to `proxy` and the file to `proxy.ts`.
import { NextRequest, NextResponse, NextFetchEvent } from 'next/server';
export function middleware(request: NextRequest, event: NextFetchEvent) {
const { pathname } = request.nextUrl;
// Never track your own API or static assets: the matcher below excludes
// most of these, this is a second-layer guard.
if (pathname.startsWith('/api/')) return NextResponse.next();
// Real client/bot IP. Prefer Cloudflare's CF-Connecting-IP (CF sets it to
// the true client) so a CF-fronted site records the real bot IP, not the
// Cloudflare edge IP; fall back to the leftmost X-Forwarded-For, then
// X-Real-IP. Matches the CF Worker's precedence.
const ip =
request.headers.get('cf-connecting-ip') ||
request.headers.get('x-forwarded-for')?.split(',')[0]?.trim() ||
request.headers.get('x-real-ip') ||
'';
// intake requires a non-empty ip; if none resolved (rare), skip the send.
if (!ip) return NextResponse.next();
// Externally-visible URL. `request.url` carries the raw Host header, and
// behind a reverse proxy — Vercel, Fly, Railway, Render, an ALB, nginx, any
// container platform — that Host is the INTERNAL address, frequently
// localhost. Arrivl stores that hostname on every event and only marks the
// install verified once it matches your real domain, so reporting the
// internal one means the site never verifies. When the proxy tells us the
// public host (leftmost value: proxies append, they don't replace), swap it
// into the origin and keep the rest of the URL untouched. No proxy in front
// → request.url is already right and is used as-is.
const fwdHost = request.headers.get('x-forwarded-host')?.split(',')[0]?.trim();
const fwdProto = request.headers.get('x-forwarded-proto')?.split(',')[0]?.trim();
const trackedUrl = fwdHost
? request.url.replace(/^https?:\/\/[^/]+/, `${fwdProto || 'https'}://${fwdHost}`)
: request.url;
const params = new URLSearchParams({
url: trackedUrl,
userAgent: request.headers.get('user-agent') || '',
ref: request.headers.get('referer') || '',
// The Accept header, raw. A real browser navigating to a page asks for
// text/html; a script wearing a browser's User-Agent usually does not.
// Empty when the request has none — Arrivl stores that as unknown.
accept: request.headers.get('accept') || '',
ip,
websiteKey: process.env.ARRIVL_WEBSITE_KEY!,
// Which snippet, which revision. Arrivl cannot update this file for you,
// so these two are how we can tell you it has gone stale.
source: 'nextjs',
sv: '3',
// WHY THERE IS NO status/bytes HERE, and why that is not an omission.
//
// Proxy (and middleware before it) runs BEFORE the route renders, and
// `NextResponse.next()` means "continue" — it is a signal, not the
// response your app went on to produce. So at this point in the request
// there is no status code and no byte count in existence to report, and
// inventing 200 would put a fabricated number on every soft-404 and every
// 5xx an AI crawler hits.
//
// `shape=pre_response` says exactly that to Arrivl, so a blank status on
// this install is reported as NOT APPLICABLE rather than as a missing
// field. If you want the real status and size, report from somewhere that
// has the response instead — a Node server's `res.on('finish')`, a
// framework "after response" hook, or an access-log shipper — and declare
// `shape=post_response` / `shape=log` there.
shape: 'pre_response',
});
// Web Bot Auth signature headers — forward raw when present (rare today).
for (const [h, p] of [['signature', 'sig'], ['signature-input', 'sigInput'], ['signature-agent', 'sigAgent']]) {
const v = request.headers.get(h);
if (v) params.set(p, v);
}
// event.waitUntil sends the request in the background WITHOUT blocking the
// response. A bare un-awaited fetch is killed on Edge/serverless (Vercel,
// Cloudflare, Netlify): the runtime freezes the function the instant the
// response is returned, so the event never leaves. waitUntil keeps it alive,
// on every platform (no @vercel/functions needed).
event.waitUntil(
fetch(`https://arrivl.ai/api/v1/intake/pageview?${params}`).catch(() => {})
);
return NextResponse.next();
}
// Track AI-discovery paths (robots.txt, llms.txt, sitemap.xml, .well-known)
// even though their extensions look static: these are the highest-signal
// hits Arrivl captures. The catch-all below then excludes other .txt/.xml/.json.
export const config = {
matcher: [
'/robots.txt',
'/llms.txt',
'/llms-full.txt',
'/sitemap.xml',
'/ai.txt',
'/sitemap-:path*',
'/.well-known/:path*',
'/((?!api|_next/static|_next/image|favicon.ico|manifest.webmanifest|.*\\.(?:png|jpg|jpeg|svg|gif|webp|ico|css|js|mjs|map|woff2?|ttf|otf|eot|txt|xml|json)$).*)',
],
};Requires ARRIVL_WEBSITE_KEY in your environment, same as above.
Values shown as ak_YOUR_WEBSITE_KEY are placeholders. Sign in and the in-dashboard version of this guide fills in your real key and per-project values for you.
Next step