""" Centralized configuration for GEO Optimizer. All shared constants (bots, schemas, scoring weights, patterns) live here so that core modules, CLI, and tests can import from a single source. """ from __future__ import annotations import os from pathlib import Path # ─── HTTP ──────────────────────────────────────────────────────────────────── USER_AGENT = "GEO-Optimizer/2.0 (https://github.com/auriti-labs/geo-optimizer-skill)" HEADERS = {"User-Agent": USER_AGENT} # Real desktop browser UA, used only to recover a resource that the CDN/WAF # blocks for the auditor's User-Agent (e.g. llms.txt behind a bot wall). It is # an honest, well-formed browser fingerprint — not a spoofed bot signature — and # is deliberately a separate, explicit constant so it is never applied as the # default fetch UA (#528 override stays independent). BROWSER_USER_AGENT = ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" ) # Process-wide User-Agent override for the generic fetch layer (#528). Resolved # once at CLI startup from --user-agent / GEO_USER_AGENT and read by # utils/http.py, utils/http_async.py and llms_generator.py via get_headers(). # Deliberately does NOT affect AI_BOTS / CITATION_BOTS-based checks (e.g. the # CDN AI-crawler probe, #225): those send a specific bot identity on purpose, # and overriding it would defeat the test. _user_agent_override: str | None = None def set_user_agent_override(user_agent: str | None) -> None: """Set (or clear, with None) the process-wide User-Agent override.""" global _user_agent_override _user_agent_override = user_agent.strip() if user_agent and user_agent.strip() else None def resolve_user_agent_override(cli_value: str | None) -> str | None: """Resolve a User-Agent override from a CLI flag or the GEO_USER_AGENT env var. The CLI flag takes precedence. Returns None if neither is set, meaning the default USER_AGENT applies. """ if cli_value and cli_value.strip(): return cli_value.strip() env_value = os.environ.get("GEO_USER_AGENT", "").strip() return env_value or None def get_headers() -> dict: """Return the current fetch headers, honoring any active User-Agent override.""" if _user_agent_override: return {"User-Agent": _user_agent_override} return dict(HEADERS) # HTTP response size limit: 10 MB (prevents DoS from huge responses) — fix #91 MAX_RESPONSE_SIZE: int = 10 * 1024 * 1024 # Maximum number of sub-sitemaps to process in a sitemap index — fix #90 MAX_SUB_SITEMAPS: int = 10 # Total URL limit extracted from all sitemaps — fix #124 (sitemap bomb) MAX_TOTAL_URLS: int = 10_000 # ─── Local history / tracking ──────────────────────────────────────────────── # Performance budget: warn if a single-page audit exceeds this threshold (#290) AUDIT_TIMEOUT_SECONDS: int = 10 GEO_OPTIMIZER_HOME = Path.home() / ".geo-optimizer" TRACKING_DB_PATH = GEO_OPTIMIZER_HOME / "tracking.db" SNAPSHOTS_DB_PATH = GEO_OPTIMIZER_HOME / "snapshots.db" DEFAULT_HISTORY_RETENTION_DAYS = 90 DEFAULT_HISTORY_LIMIT = 12 DEFAULT_SNAPSHOT_LIMIT = 20 # ─── Passive AI visibility monitoring ──────────────────────────────────────── MONITOR_SCORING = { "citation_bot_access": 20, "user_fetch_access": 10, "llms_readiness": 15, "ai_discovery_readiness": 15, "entity_strength": 15, "trust_strength": 15, "momentum": 10, } MONITOR_BANDS = { "strong": (80, 100), "visible": (60, 79), "emerging": (35, 59), "low": (0, 34), } # ─── AI bots — 3-tier classification (training/search/user) ────────────────── # # Training: crawl to train models (less critical for direct visibility) # Search: cite the site in AI responses (highest priority for GEO) # User: on-demand fetch when a user asks about a specific URL AI_BOTS = { # ── OpenAI ────────────────────────────────────────────────────────────── "GPTBot": "OpenAI (ChatGPT training)", "OAI-SearchBot": "OpenAI (ChatGPT search citations)", "ChatGPT-User": "OpenAI (ChatGPT on-demand fetch)", # ── Anthropic ─────────────────────────────────────────────────────────── # anthropic-ai/claude-web removed (#512): not listed in Anthropic's current # published crawler docs (support.claude.com), which name exactly these three. "ClaudeBot": "Anthropic (Claude training)", "Claude-SearchBot": "Anthropic (Claude search citations)", "Claude-User": "Anthropic (Claude on-demand fetch)", # ── Perplexity ────────────────────────────────────────────────────────── "PerplexityBot": "Perplexity AI (index builder)", "Perplexity-User": "Perplexity (citation fetch on-demand)", # ── Google ────────────────────────────────────────────────────────────── # Googlebot (#512): the same crawler that feeds Search also feeds AI # Overviews — Google's own docs state Google-Extended is a robots.txt # token layered on Googlebot's data, not a separate fetching agent, and # controls only Gemini/Vertex training, not AI Overviews eligibility. "Googlebot": "Google (Search + AI Overviews)", "Google-Extended": "Google (Gemini/Vertex training opt-out — not a crawler)", "Google-CloudVertexBot": "Google (Vertex AI)", # ── Microsoft ─────────────────────────────────────────────────────────── "Bingbot": "Microsoft (Bing/Copilot search)", # ── Apple ─────────────────────────────────────────────────────────────── "Applebot-Extended": "Apple (AI training)", # ── Other ─────────────────────────────────────────────────────────────── "cohere-ai": "Cohere (language models)", "DuckAssistBot": "DuckDuckGo AI", "Bytespider": "ByteDance/TikTok AI", "meta-externalagent": "Meta AI (Facebook/Instagram AI)", # ── Meta (expanded) ───────────────────────────────────────────────────── "Meta-ExternalFetcher": "Meta (content fetch on-demand)", "facebookexternalhit": "Meta (social preview + AI)", # ── Amazon ────────────────────────────────────────────────────────────── "Amazonbot": "Amazon (Alexa/search AI)", # ── Allen Institute ───────────────────────────────────────────────────── "AI2Bot": "Allen Institute (AI research)", "AI2Bot-Dolma": "Allen Institute (Dolma dataset)", # ── xAI ──────────────────────────────────────────────────────────────── "xAI-Bot": "xAI (Grok search citations)", # ── Apple (general) ──────────────────────────────────────────────────── "Applebot": "Apple (general web crawl + Siri AI)", # ── Huawei ───────────────────────────────────────────────────────────── "PetalBot": "Huawei (PetalSearch AI, EU/Asia)", # ── You.com ───────────────────────────────────────────────────────────── "YouBot": "You.com AI search", # ── Common Crawl ──────────────────────────────────────────────────────── "CCBot": "Common Crawl (used by many AI labs)", } # 3-tier classification — bots grouped by function BOT_TIERS = { "training": { "GPTBot", "ClaudeBot", "Google-Extended", "Google-CloudVertexBot", "Applebot-Extended", "cohere-ai", "Bytespider", "meta-externalagent", "PetalBot", "AI2Bot", "AI2Bot-Dolma", "CCBot", }, "search": { "OAI-SearchBot", "Claude-SearchBot", "PerplexityBot", "Googlebot", "Applebot", "Bingbot", "DuckAssistBot", "YouBot", "Amazonbot", "xAI-Bot", }, "user": { "ChatGPT-User", "Claude-User", "Perplexity-User", "Meta-ExternalFetcher", "facebookexternalhit", }, } # Critical citation bots (search-tier bots that actually drive AI citations — # #512: matches the "AI search crawlers" set, not the training-only crawlers # that happen to share a vendor. ClaudeBot is training-only per Anthropic's # current docs, so it is excluded here even though it is Anthropic's bot). CITATION_BOTS = {"OAI-SearchBot", "Claude-SearchBot", "PerplexityBot", "Googlebot", "Applebot"} # Human-readable bot labels for user-facing recommendation messages ROBOTS_KEY_BOTS_DISPLAY: str = "GPTBot, ClaudeBot, PerplexityBot" CITATION_BOTS_DISPLAY: str = "OAI-SearchBot, Claude-SearchBot, PerplexityBot, Googlebot, Applebot" # ─── Brand normalization ────────────────────────────────────────────────────── # Legal suffixes stripped from brand names before comparison (#397). # Only removed when they appear at the END of the name (after stripping punctuation/spaces). # Lowercase, matched against the lowercased trailing token(s). BRAND_LEGAL_SUFFIXES: frozenset = frozenset( { "inc", "inc.", "incorporated", "ltd", "ltd.", "limited", "llc", "l.l.c.", "corp", "corp.", "corporation", "gmbh", "g.m.b.h.", "s.r.l.", "srl", "s.p.a.", "spa", "s.a.", "sa", "ag", "co", "co.", "plc", "pty", "pty.", "bv", "b.v.", "nv", "n.v.", } ) # ─── Schema types ──────────────────────────────────────────────────────────── # All schema.org Article subtypes that count as Article for GEO scoring # Includes direct subclasses per schema.org hierarchy (#392) ARTICLE_TYPES: frozenset[str] = frozenset( { "Article", "BlogPosting", "NewsArticle", "TechArticle", "ScholarlyArticle", } ) # schema.org Organization subtypes that count as Organization for GEO scoring # (entity/trust signals, contact-info validation). Same fix shape as # ARTICLE_TYPES/#392: a node typed "LocalBusiness" (or one of its own common # subtypes) IS an Organization per schema.org's hierarchy, but was previously # only matched by the literal string "Organization" — which most real-world # small-business sites never use directly, since LocalBusiness and its # subtypes are schema.org's own recommended, more specific types for exactly # that audience. ORGANIZATION_TYPES: frozenset[str] = frozenset( { "Organization", # Direct schema.org subtypes of Organization "LocalBusiness", "Corporation", "EducationalOrganization", "GovernmentOrganization", "MedicalOrganization", "NGO", "NewsMediaOrganization", "OnlineBusiness", "PerformingGroup", "SportsOrganization", # Common LocalBusiness subtypes used directly as @type "Store", "Restaurant", "FoodEstablishment", "ProfessionalService", "HomeAndConstructionBusiness", "AutomotiveBusiness", "MedicalBusiness", "Dentist", "Attorney", "LegalService", "FinancialService", "RealEstateAgent", "LodgingBusiness", "Hotel", "HealthAndBeautyBusiness", "EntertainmentBusiness", "GovernmentOffice", "Library", } ) VALUABLE_SCHEMAS = [ "WebSite", "WebApplication", "FAQPage", "Article", "BlogPosting", "NewsArticle", "TechArticle", "ScholarlyArticle", "HowTo", "Recipe", "Product", "Organization", "Person", "BreadcrumbList", ] # Required fields for each schema.org type (keys are lowercase) SCHEMA_ORG_REQUIRED = { "website": ["@context", "@type", "url", "name"], "webpage": ["@context", "@type", "url", "name"], "organization": ["@context", "@type", "name", "url"], "person": ["@context", "@type", "name"], "faqpage": ["@context", "@type", "mainEntity"], "article": ["@context", "@type", "headline", "author"], "breadcrumblist": ["@context", "@type", "itemListElement"], "product": ["@context", "@type", "name", "description"], "localbusiness": ["@context", "@type", "name", "address"], "webapplication": ["@context", "@type", "name", "url"], } SCHEMA_TEMPLATES = { "website": { "@context": "https://schema.org", "@type": "WebSite", "name": "{{name}}", "url": "{{url}}", "description": "{{description}}", "potentialAction": { "@type": "SearchAction", "target": { "@type": "EntryPoint", "urlTemplate": "{{url}}/search?q={search_term_string}", }, "query-input": "required name=search_term_string", }, }, "webapp": { "@context": "https://schema.org", "@type": "WebApplication", "name": "{{name}}", "url": "{{url}}", "description": "{{description}}", "applicationCategory": "UtilityApplication", "operatingSystem": "Web", "browserRequirements": "Requires JavaScript", "offers": {"@type": "Offer", "price": "0", "priceCurrency": "USD"}, "author": {"@type": "Organization", "name": "{{author}}"}, }, "faq": { "@context": "https://schema.org", "@type": "FAQPage", "mainEntity": [], }, "article": { "@context": "https://schema.org", "@type": "Article", "headline": "{{title}}", "description": "{{description}}", "url": "{{url}}", # image field required for Google Rich Results (#112) "image": "{{image_url}}", "datePublished": "{{date_published}}", "dateModified": "{{date_modified}}", "author": {"@type": "Person", "name": "{{author}}"}, "publisher": { "@type": "Organization", "name": "{{publisher}}", "logo": {"@type": "ImageObject", "url": "{{logo_url}}"}, }, }, "organization": { "@context": "https://schema.org", "@type": "Organization", "name": "{{name}}", "url": "{{url}}", "description": "{{description}}", # logo must be ImageObject, not URL string (#113) "logo": {"@type": "ImageObject", "url": "{{logo_url}}"}, # sameAs is the most important signal for brand_kg_readiness (3pt — #398) # Placeholders use authoritative domains from SAMEAS_AUTHORITATIVE_DOMAINS "sameAs": [ "https://www.linkedin.com/company/YOUR_COMPANY", "https://github.com/YOUR_ORG", "https://twitter.com/YOUR_HANDLE", ], }, "breadcrumb": { "@context": "https://schema.org", "@type": "BreadcrumbList", "itemListElement": [{"@type": "ListItem", "position": 1, "name": "Home", "item": "{{url}}"}], }, # HowTo, Review and Product close the detect->generate gap: audit_schema.py already # scores has_howto/has_product, and Review/AggregateRating is one of the more # citation-relevant types per GEO research, but `geo schema --type` had no template # for any of the three — a user told "you're missing HowTo schema" had no way to ask # the tool that told them so to generate it. "howto": { "@context": "https://schema.org", "@type": "HowTo", "name": "{{title}}", "description": "{{description}}", "image": "{{image_url}}", "totalTime": "{{total_time}}", "step": [ {"@type": "HowToStep", "name": "{{step_1_name}}", "text": "{{step_1_text}}"}, ], }, "review": { "@context": "https://schema.org", "@type": "Review", "itemReviewed": {"@type": "Thing", "name": "{{name}}"}, # ratingValue/bestRating as strings: schema.org accepts Number or Text, and a # template placeholder is text until the user fills it in — matches how every # other numeric-looking field in this file's templates is handled. "reviewRating": {"@type": "Rating", "ratingValue": "{{rating_value}}", "bestRating": "5"}, "author": {"@type": "Person", "name": "{{author}}"}, "reviewBody": "{{review_body}}", }, "product": { "@context": "https://schema.org", "@type": "Product", "name": "{{name}}", "description": "{{description}}", "image": "{{image_url}}", "brand": {"@type": "Brand", "name": "{{author}}"}, "offers": { "@type": "Offer", "price": "{{price}}", "priceCurrency": "USD", "availability": "https://schema.org/InStock", }, }, } # ─── llms.txt patterns ────────────────────────────────────────────────────── CATEGORY_PATTERNS = [ (r"/blog/", "Blog & Articles"), (r"/article/", "Articles"), (r"/articles/", "Articles"), (r"/post/", "Posts"), (r"/news/", "News"), (r"/finance/", "Finance Tools"), (r"/health/", "Health & Wellness"), (r"/math/", "Math"), (r"/calcul", "Calculators"), (r"/tool/", "Tools"), (r"/tools/", "Tools"), (r"/app/", "Applications"), (r"/docs?/", "Documentation"), (r"/guide/", "Guides"), (r"/tutorial/", "Tutorials"), (r"/tutorials/", "Tutorials"), # Patterns with slash to avoid false positives (#117) # /product → /production-process, /service → /service-terms (r"/products/", "Products"), (r"/product/", "Products"), (r"/services/", "Services"), (r"/service/", "Services"), # New categories (#118) (r"/faq/", "FAQ"), (r"/faqs/", "FAQ"), (r"/pricing/", "Pricing"), (r"/price/", "Pricing"), (r"/portfolio/", "Portfolio"), (r"/case-stud", "Case Studies"), (r"/support/", "Support"), (r"/help/", "Support"), (r"/team/", "Team"), (r"/about-us(?:/|$)", "About"), (r"/about(?:/|$)", "About"), (r"/careers/", "Careers"), (r"/jobs/", "Careers"), (r"/contact", "Contact"), (r"/privacy", "Privacy & Legal"), (r"/terms", "Terms"), ] SKIP_PATTERNS = [ r"/wp-", r"/admin", r"/login", r"/logout", r"/register", r"/cart", r"/checkout", r"/account", r"/user/", r"\.(xml|json|rss|atom|pdf|jpg|png|css|js)$", r"/tag/", r"/category/\w+/page/", r"/page/\d+", # Additional skip patterns (#118) r"/feed/", r"/author/", r"/amp/", r"/api/", r"/wp-json/", ] # llms.txt section ordering SECTION_PRIORITY_ORDER = [ "Tools", "Calculators", "Finance Tools", "Health & Wellness", "Math", "Applications", "Main Pages", "Documentation", "Guides", "Tutorials", "Blog & Articles", "Articles", "Posts", "News", "Products", "Services", "FAQ", "Pricing", "Portfolio", "Case Studies", "Support", "Team", "Careers", "About", "Contact", "Other", "Privacy & Legal", "Terms", ] OPTIONAL_CATEGORIES = {"Privacy & Legal", "Terms", "Contact", "Other"} # ─── Scoring weights ───────────────────────────────────────────────────────── SCORING = { # robots.txt — 18 points (was 20) "robots_found": 5, "robots_citation_ok": 13, # was 15 # robots_some_allowed: removed from dict, now in ROBOTS_PARTIAL_SCORE (fix #332) # llms.txt — 18 points (was 20) — graduated quality + blockquote v2 "llms_found": 5, # was 6 — 1 point moved to llms_blockquote (#39) "llms_h1": 2, # was 3 "llms_blockquote": 1, # #39: blockquote description present "llms_sections": 2, # was 4 "llms_links": 2, # was 3 "llms_depth": 2, # NEW: word_count >= 1000 "llms_depth_high": 2, # NEW: word_count >= 5000 "llms_full": 2, # NEW: has llms-full.txt # Schema JSON-LD — 16 points (was 25) — any valid type + sameAs + richness "schema_any_valid": 2, # any valid JSON-LD schema found (was 5, reduced for richness) "schema_richness": 3, # NEW: schema with 5+ relevant attributes (Growth Marshal 2026) "schema_faq": 3, # was 5 — reduced, migrated to brand_topic_authority "schema_article": 3, # was 4 "schema_organization": 3, # was 3 "schema_website": 2, # was 3 "schema_sameas": 0, # was 3, migrated to brand KG — kept at 0 for backward compat # Meta tags — 14 points "meta_title": 5, "meta_description": 2, "meta_canonical": 3, "meta_og": 4, # Content quality — 12 points (was 15) — structure checks "content_h1": 2, # was 3 "content_numbers": 1, # was 2 "content_links": 1, # was 2 "content_word_count": 2, # was 4 "content_heading_hierarchy": 2, # NEW: has H2 + H3 in correct hierarchy "content_lists_or_tables": 2, # NEW: has