GEO CHEAT SHEET SCANNER - RETIRED SOURCE ARCHIVE Preserved October 7, 2026. Do not deploy this URL fetcher. Its hostname checks do not validate DNS targets, so it could reach private addresses. CribThis uses a browser-only pasted-HTML checklist instead. The historical scoring weights are heuristics, not citation probabilities. ===== src/routes/scan.tsx ===== import { createFileRoute } from "@tanstack/react-router"; import { useState } from "react"; import { SiteHeader, SiteFooter } from "@/components/site/SiteHeader"; import type { ScanResult } from "@/lib/scan-core"; export const Route = createFileRoute("/scan")({ component: ScanPage, head: () => ({ meta: [ { title: "Scan my site — GEO Cheat Sheet" }, { name: "description", content: "A 15-second, receipt-backed check on whether your page is set up to be cited by ChatGPT, Claude, Perplexity, and Google AIO." }, { property: "og:title", content: "Is your page cite-worthy? Free GEO scan" }, { property: "og:description", content: "Instant, receipt-backed audit against the 12 factors that predict AI citation. Share the report with your team." }, { name: "twitter:title", content: "Is your page cite-worthy? Free GEO scan" }, { name: "twitter:description", content: "Instant, receipt-backed audit against the 12 factors that predict AI citation." }, { property: "og:url", content: "https://geocheatsheet.com/scan" }, ], links: [{ rel: "canonical", href: "https://geocheatsheet.com/scan" }], }), }); const CATEGORY_TAGLINE: Record = { technical: "Fix the machine-readable layer first — indexing and schema outweigh content polish here.", content: "The page is reachable but thin. Add original quotes, stats, and length before spending on distribution.", authority: "Retrieval sees the page; it just doesn't trust it yet. Named authors and co-citations are the lever.", distribution: "The page is strong on its own — the ceiling now is who links to it and who mentions it.", }; function ScanPage() { const [url, setUrl] = useState(""); const [loading, setLoading] = useState(false); const [error, setError] = useState(null); const [result, setResult] = useState(null); const submit = async (e: React.FormEvent) => { e.preventDefault(); if (!url.trim()) return; setLoading(true); setError(null); setResult(null); try { const res = await fetch("/api/public/scan", { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ url: url.trim() }), }); const payload = (await res.json().catch(() => null)) as ScanResult | { error?: string } | null; if (!res.ok || !payload || "error" in payload) { setError((payload as { error?: string } | null)?.error ?? `Scan failed (HTTP ${res.status})`); } else { setResult(payload as ScanResult); } } catch (e) { setError((e as Error).message ?? "Scan failed"); } finally { setLoading(false); } }; return (
Free tool

Is your page cite-worthy?

Paste any public URL and get a 12-factor scan against the receipts on this sheet. Directional, not exhaustive — good enough to bring to your team, honest enough to be useful.

setUrl(e.target.value)} placeholder="https://yourdomain.com/page-to-check" className="flex-1 px-3 py-2.5 text-sm bg-card border border-border focus:outline-none focus:ring-2 focus:ring-ring" />
{error &&
{error}
}
{result && ( <>
Scan of {result.finalUrl}
Cite-worthiness
= 70 ? "text-signal-high" : result.score >= 40 ? "text-signal-med" : "text-signal-low"}`}> {result.score}
out of 100
Priority area
{result.category}

{CATEGORY_TAGLINE[result.category]}

Findings
    {result.findings.map((f) => (
  • {f.passed ? "✓" : "✗"}
    {f.label}
    {f.detail}
    w {f.weight}
  • ))}
Recommended next step
{result.recommendation === "brand-receipts" ? ( ) : ( )}

Want to share this with your team? Just send them this page. Scans are stateless and don't collect the URL you tested.

)} {!result && !loading && (
What we check

Every factor here maps to a receipt on the matrix — no folk wisdom. The weights reflect what the underlying research says actually moves citation probability.

  • · descriptive <title>
  • · meta description
  • · single H1
  • · body length
  • · direct quotes
  • · numeric stats
  • · JSON-LD schema
  • · canonical URL
  • · named author
  • · Open Graph tags
  • · indexable (no noindex)
  • · HTTPS
)}
); } function RecCard({ href, title, body, cta }: { href: string; title: string; body: string; cta: string }) { return (
{title}

{body}

{cta}
); } function SecondaryCard({ href, title, body }: { href: string; title: string; body: string }) { return (
{title}
{body}
); } ===== src/lib/scan-core.ts ===== export type ScanFinding = { key: string; label: string; passed: boolean; detail: string; weight: number; }; export type ScanResult = { url: string; finalUrl: string; status: number; title: string | null; description: string | null; wordCount: number; score: number; findings: ScanFinding[]; recommendation: "brand-receipts" | "im-cited"; category: "technical" | "content" | "authority" | "distribution"; }; function stripTags(html: string): string { return html .replace(//gi, " ") .replace(//gi, " ") .replace(/<[^>]+>/g, " ") .replace(/\s+/g, " ") .trim(); } function pick(html: string, re: RegExp): string | null { const m = html.match(re); return m ? m[1].trim() : null; } const ALLOWED_PORTS = new Set(["", "80", "443"]); function ipv4ToParts(host: string): number[] | null { const m = host.match(/^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/); if (!m) return null; const parts = m.slice(1).map(Number); return parts.every((n) => n >= 0 && n <= 255) ? parts : null; } /** Blocks loopback, private, link-local, CGNAT and other non-public targets. */ function isBlockedHost(hostname: string): boolean { const host = hostname.toLowerCase().replace(/^\[|\]$/g, ""); if (!host) return true; // Hostnames that never point at the public internet if ( host === "localhost" || host.endsWith(".localhost") || host.endsWith(".local") || host.endsWith(".internal") || host.endsWith(".lan") || !host.includes(".") ) { // Bare single-label hosts (e.g. "intranet") are internal by definition if (!host.includes(".") || host === "localhost") return true; return true; } // Cloud metadata endpoints if (host === "metadata.google.internal" || host === "metadata") return true; const v4 = ipv4ToParts(host); if (v4) { const [a, b] = v4; if (a === 0 || a === 10 || a === 127) return true; if (a === 169 && b === 254) return true; // link-local + 169.254.169.254 if (a === 172 && b >= 16 && b <= 31) return true; if (a === 192 && b === 168) return true; if (a === 100 && b >= 64 && b <= 127) return true; // CGNAT if (a >= 224) return true; // multicast + reserved return false; } // IPv6 literals: only allow globally routable 2000::/3 if (host.includes(":")) { if (host === "::1" || host === "::") return true; if (/^(fc|fd|fe80|ff)/.test(host)) return true; if (/^::ffff:/.test(host)) return true; // IPv4-mapped return !/^[23]/.test(host); } return false; } export function normalizeUrl(input: string): string { const trimmed = (input ?? "").trim(); if (!trimmed) throw new Error("URL required"); let u: URL; try { u = new URL(trimmed.startsWith("http") ? trimmed : `https://${trimmed}`); } catch { throw new Error("Enter a valid URL"); } if (u.protocol !== "http:" && u.protocol !== "https:") throw new Error("Only http(s) URLs"); if (!ALLOWED_PORTS.has(u.port)) throw new Error("Only standard web ports (80/443) can be scanned"); if (u.username || u.password) throw new Error("URLs with embedded credentials are not allowed"); if (isBlockedHost(u.hostname)) throw new Error("That address isn't a public website"); return u.toString(); } export async function performScan(rawUrl: string): Promise { const url = normalizeUrl(rawUrl); // Follow redirects manually so every hop is re-validated against the blocklist. let response: Response; let current = url; try { for (let hop = 0; ; hop++) { if (hop > 5) throw new Error("Too many redirects"); const res = await fetch(current, { redirect: "manual", headers: { "User-Agent": "Mozilla/5.0 (compatible; GEOCheatSheetBot/1.0; +https://geocheatsheet.com/scan)", Accept: "text/html,application/xhtml+xml", }, signal: AbortSignal.timeout(15000), }); const location = res.status >= 300 && res.status < 400 ? res.headers.get("location") : null; if (!location) { response = res; break; } current = normalizeUrl(new URL(location, current).toString()); } } catch (e) { throw new Error(`Could not reach ${url}: ${(e as Error).message}`); } const finalUrl = current; const html = await response.text(); const title = pick(html, /]*>([\s\S]*?)<\/title>/i); const description = pick(html, /]+name=["']description["'][^>]+content=["']([^"']+)["']/i) ?? pick(html, /]+content=["']([^"']+)["'][^>]+name=["']description["']/i); const h1 = pick(html, /]*>([\s\S]*?)<\/h1>/i); const canonical = pick(html, /]+rel=["']canonical["'][^>]+href=["']([^"']+)["']/i); const authorMeta = pick(html, /]+name=["']author["'][^>]+content=["']([^"']+)["']/i); const bodyText = stripTags(html); const wordCount = bodyText.split(/\s+/).filter(Boolean).length; const quoteMatches = html.match(/“|”|["'"'']/g) ?? []; const numberMatches = bodyText.match(/\b\d[\d,\.]*\s*(%|percent|million|billion|thousand|k|users|respondents)\b/gi) ?? []; const hasSchema = /]+type=["']application\/ld\+json["']/i.test(html); const hasFaqSchema = /"@type"\s*:\s*"FAQPage"/i.test(html); const hasOg = /]+property=["']og:/i.test(html); const hasStructuredAuthor = /"@type"\s*:\s*"Person"/i.test(html) || /"author"\s*:/i.test(html); const hasNoIndex = /]+name=["']robots["'][^>]+noindex/i.test(html); const hasHttps = finalUrl.startsWith("https://"); const findings: ScanFinding[] = [ { key: "title", label: "Descriptive tag", passed: !!title && title.length >= 20 && title.length <= 70, detail: title ? `“${title}” (${title.length} chars)` : "Missing", weight: 6 }, { key: "description", label: "Meta description present", passed: !!description && description.length >= 50, detail: description ? `${description.length} chars` : "Missing", weight: 4 }, { key: "h1", label: "Single H1 with the topic", passed: !!h1, detail: h1 ? h1.slice(0, 80) : "Missing", weight: 5 }, { key: "wordcount", label: "Substantive body (≥ 600 words)", passed: wordCount >= 600, detail: `${wordCount} words`, weight: 8 }, { key: "quotes", label: "Direct quotes in the copy", passed: quoteMatches.length >= 6, detail: `${quoteMatches.length} quote marks detected`, weight: 8 }, { key: "stats", label: "Numeric statistics in the copy", passed: numberMatches.length >= 2, detail: `${numberMatches.length} number/stat phrases`, weight: 8 }, { key: "schema", label: "JSON-LD structured data", passed: hasSchema, detail: hasSchema ? (hasFaqSchema ? "Present · includes FAQPage" : "Present") : "Missing", weight: 3 }, { key: "canonical", label: "Canonical URL", passed: !!canonical, detail: canonical ?? "Missing", weight: 4 }, { key: "author", label: "Named author signal", passed: !!(authorMeta || hasStructuredAuthor), detail: authorMeta ?? (hasStructuredAuthor ? "Structured author present" : "Missing"), weight: 6 }, { key: "og", label: "Open Graph / social preview tags", passed: hasOg, detail: hasOg ? "Present" : "Missing", weight: 2 }, { key: "indexable", label: "Indexable (no noindex)", passed: !hasNoIndex, detail: hasNoIndex ? "Blocked from index" : "OK", weight: 8 }, { key: "https", label: "Served over HTTPS", passed: hasHttps, detail: hasHttps ? "OK" : "Insecure", weight: 3 }, ]; const totalWeight = findings.reduce((a, f) => a + f.weight, 0); const hit = findings.reduce((a, f) => a + (f.passed ? f.weight : 0), 0); const score = Math.round((hit / totalWeight) * 100); const personalSignals = /\b(about me|my (blog|newsletter|work)|personal (site|brand)|@[A-Za-z0-9_]+)\b/i.test(bodyText.slice(0, 4000)); const recommendation = personalSignals ? "im-cited" : "brand-receipts"; const category: ScanResult["category"] = !hasSchema || !canonical || hasNoIndex || !hasHttps ? "technical" : wordCount < 600 || quoteMatches.length < 6 || numberMatches.length < 2 ? "content" : !(authorMeta || hasStructuredAuthor) ? "authority" : "distribution"; return { url, finalUrl, status: response.status, title, description, wordCount, score, findings, recommendation, category, }; } ===== src/routes/api/public/scan.ts ===== import { createFileRoute } from "@tanstack/react-router"; import { performScan } from "@/lib/scan-core"; function cors(res: Response): Response { res.headers.set("Access-Control-Allow-Origin", "*"); res.headers.set("Access-Control-Allow-Methods", "POST, OPTIONS"); res.headers.set("Access-Control-Allow-Headers", "content-type"); return res; } export const Route = createFileRoute("/api/public/scan")({ server: { handlers: { OPTIONS: async () => cors(new Response(null, { status: 204 })), POST: async ({ request }) => { let body: unknown; try { body = await request.json(); } catch { return cors(Response.json({ error: "Invalid JSON body" }, { status: 400 })); } const url = (body as { url?: unknown } | null)?.url; if (typeof url !== "string" || !url.trim()) { return cors(Response.json({ error: "URL is required" }, { status: 400 })); } try { const result = await performScan(url); return cors(Response.json(result)); } catch (e) { const message = e instanceof Error ? e.message : "Scan failed"; return cors(Response.json({ error: message }, { status: 400 })); } }, }, }, });