{
  "$schema": "https://json-schema.org/draft/2020-12/schema",
  "profile": "HTML&HTML Open Website AI Readiness Audit Profile",
  "profileId": "hh-owairp",
  "version": "3.0.0",
  "status": "public-product-profile",
  "published": "2026-09-06",
  "canonical": "https://htmlandhtml.com/standard/",
  "disclaimer": "This is an openly published HTML&HTML product audit profile, not an IETF, W3C, Google, OpenAI, Anthropic or other third-party standard.",
  "principles": [
    "Measured facts, normative standards, vendor guidance, proposals and heuristics must remain distinguishable.",
    "Unknown or unavailable evidence must not be forced into pass or fail.",
    "A score describes measured technical conditions only; it does not guarantee ranking, traffic, revenue or AI citation.",
    "Public scanning must not invent private source-code ownership or file paths.",
    "Security boundaries must fail closed for private, local, reserved or credential-bearing targets.",
    "Experimental agent surfaces must not be published solely to improve a score.",
    "The Intelligence Audit layer is decision support and never mutates the canonical 18-engine overall score."
  ],
  "sourceClasses": [
    {
      "id": "OFFICIAL_STANDARD",
      "meaning": "Normative or standards-track specification."
    },
    {
      "id": "OFFICIAL_VENDOR",
      "meaning": "Current first-party product or crawler guidance."
    },
    {
      "id": "PROPOSAL",
      "meaning": "Emerging convention that is not treated as a mandatory standard."
    },
    {
      "id": "MEASURED",
      "meaning": "Directly observed HTTP, HTML, header, robots or link evidence."
    },
    {
      "id": "INTERNAL_HEURISTIC",
      "meaning": "HTML&HTML decision-support rule where no normative threshold exists."
    },
    {
      "id": "EXPERIMENTAL",
      "meaning": "Early or optional capability with limited maturity or adoption."
    }
  ],
  "confidenceLevels": [
    "confirmed",
    "strong",
    "probable",
    "requires-source-verification"
  ],
  "ruleStates": [
    "PASS",
    "FAIL",
    "UNKNOWN",
    "NOT_MEASURED",
    "REQUIRES_CONTEXT"
  ],
  "engines": [
    {
      "id": "crawl",
      "name": "Crawl & Index",
      "overallWeight": 12,
      "purpose": "Public reachability, redirects, robots, sitemap, indexability and page discovery."
    },
    {
      "id": "technical",
      "name": "Technical SEO",
      "overallWeight": 14,
      "purpose": "Titles, descriptions, headings, canonicals, duplicate and route-level technical signals."
    },
    {
      "id": "ai",
      "name": "AI / GEO Access",
      "overallWeight": 12,
      "purpose": "Search, retrieval, Google Preferred Sources, and AI crawler policy signals without conflating search access with training controls."
    },
    {
      "id": "llms",
      "name": "llms.txt v2",
      "overallWeight": 6,
      "purpose": "Proposal-format, link reachability and discovery relations such as describedby and Markdown alternates."
    },
    {
      "id": "schema",
      "name": "Structured Data",
      "overallWeight": 8,
      "purpose": "JSON-LD parseability, entity types and visible-identity consistency signals."
    },
    {
      "id": "performance",
      "name": "Performance Hygiene",
      "overallWeight": 10,
      "purpose": "Observable HTML/script/render-blocking hygiene; field CWV remain NOT_MEASURED without reliable field or lab data."
    },
    {
      "id": "accessibility",
      "name": "Accessibility",
      "overallWeight": 9,
      "purpose": "Document language, image alternatives, programmatic labels and accessible control names."
    },
    {
      "id": "security",
      "name": "Security Baseline",
      "overallWeight": 10,
      "purpose": "HTTPS, browser security headers, mixed-content and insecure-form baseline controls."
    },
    {
      "id": "trust",
      "name": "Content Trust",
      "overallWeight": 7,
      "purpose": "Organization identity, contact, privacy, terms and editorial accountability signals."
    },
    {
      "id": "agent",
      "name": "Agent Readiness",
      "overallWeight": 3,
      "purpose": "Machine-readable and programmatic discovery surfaces while keeping experimental protocols explicitly optional."
    },
    {
      "id": "conversion",
      "name": "Conversion",
      "overallWeight": 4,
      "purpose": "Visibility and continuity of public next-step/contact/form flows without changing business logic."
    },
    {
      "id": "links",
      "name": "Link Integrity",
      "overallWeight": 5,
      "purpose": "Real HTTP probing of owned internal destinations and unnecessary redirects."
    }
  ],
  "intelligenceLayer": {
    "classification": "NON_SCORING_INTELLIGENCE_LAYER",
    "analysisCount": 13,
    "readinessLensCount": 7,
    "readinessLenses": [
      "SEO",
      "GEO",
      "AEO",
      "LLMO",
      "AAO",
      "RAG",
      "E-E-A-T"
    ],
    "analyses": [
      "intent_cannibalization",
      "information_gain",
      "answer_extractability",
      "entity_graph_integrity",
      "freshness_integrity",
      "render_parity",
      "llm_knowledge_surface",
      "internal_link_semantic_alignment",
      "orphan_pages",
      "discovery_path",
      "indexnow_readiness",
      "structured_graph_consistency",
      "codebase_seo_governance"
    ],
    "informationGainBoundary": "Public within-site differentiation signals only; not a reproduction of a Google ranking score or whole-web novelty model.",
    "renderParityBoundary": "NOT_MEASURED without a controlled browser-render snapshot paired with raw HTML.",
    "codebaseGovernanceBoundary": "REQUIRES_CONTEXT and activates only with authorized source/repository context.",
    "indexNowBoundary": "NOT_MEASURED unless configuration or verified public evidence is available."
  },
  "scanBoundaries": {
    "maxPublicHtmlPages": 50,
    "maxLinkProbes": 30,
    "privateOrReservedTargets": "reject",
    "credentialBearingUrls": "reject",
    "nonStandardPorts": "reject",
    "redirectPivotToPrivateNetwork": "reject",
    "fieldCoreWebVitalsWithoutReliableData": "NOT_MEASURED"
  },
  "commercialBoundary": {
    "free": "Evidence, scores, intelligence signals and priorities",
    "paid": "$99 Full Site Fix Mandate with automated code package and 30-day re-scan",
    "implementationWithoutEntitlement": "fail-closed"
  },
  "officialSourceRegistry": "https://htmlandhtml.com/sources.json",
  "methodology": "https://htmlandhtml.com/methodology.html",
  "validator": "https://htmlandhtml.com/",
  "engineContract": {
    "name": "Engine V3 Deterministic Chain",
    "moduleCount": 18,
    "controlCount": 105,
    "qualityGateRange": "G0-G9",
    "qualityGateCount": 10,
    "randomness": "none",
    "auditTrailFields": [
      "score",
      "confidence",
      "rawData",
      "ruleChain",
      "computationSteps",
      "checksum",
      "executionMs"
    ]
  },
  "machineSurfaceContract": {
    "rootLlmsTxt": 1,
    "maxPageLevelMarkdownManifests": 30,
    "llmsTxtClassification": "PROPOSAL",
    "googleSearchImpact": "NONE_DOCUMENTED_BY_GOOGLE"
  },
  "executionContract": {
    "scannerPhases": 8,
    "readinessDisciplines": [
      "SEO",
      "GEO",
      "AEO",
      "LLMO",
      "AAO",
      "RAG",
      "E-E-A-T"
    ],
    "neutralQueryMaximum": 3,
    "providerMaximum": 3,
    "providerSurfaces": [
      "OpenAI Responses API + web search",
      "Perplexity Sonar API web-search surface",
      "Gemini API + Google Search grounding"
    ],
    "maximumObservationsPerRun": 9,
    "promptMeasurementBoundary": "PAID_AND_PROVIDER_CONFIGURATION_REQUIRED; API/search-grounded surfaces are not identical to consumer UI results",
    "arrRiskClassification": "SCENARIO_ESTIMATE_NOT_MEASURED_REVENUE",
    "implementationPackage": "VERSIONED ZIP; EXACT FILE COUNT VARIES WITH EVIDENCED FINDINGS AND UP TO 30 PAGE-LEVEL MACHINE-SURFACE MANIFESTS"
  },
  "measurementTaxonomy": {
    "SEO": "Measured crawl/index/technical signals plus Google vendor guidance; no ranking guarantee.",
    "GEO": "Website-side source-readiness and generative-search eligibility signals; model recommendation is not measured.",
    "AEO": "Question/answer structure and answer-extractability heuristics; FAQ rich-result eligibility is not claimed.",
    "LLMO": "Machine-readable surfaces and llms.txt proposal checks; Google ranking impact is explicitly none.",
    "AAO": "Explicit agent/API/tool discovery surfaces only; experimental protocols remain EXPERIMENTAL.",
    "RAG": "Retrieval/chunking/readability heuristics; internal model embeddings, rerankers and vector scores are not observed.",
    "E-E-A-T": "Public trust, identity and accountability evidence; not a reproduction of a Google ranking score."
  },
  "epistemicPolicy": {
    "unmeasuredSignals": "NOT_MEASURED_AND_EXCLUDED_FROM_SCORE",
    "privateCodeSignals": "REQUIRES_CONTEXT",
    "defaultRuleSourceClass": "INTERNAL_HEURISTIC",
    "vendorClaimsRequireSourceId": true,
    "modelInternalClaimsFromPublicHtml": "FORBIDDEN"
  },
  "advancedBlackBoxRiskLayer": {
    "classification": "NON_SCORING_ADVANCED_BLACKBOX_RISK_LAYER",
    "analysisCount": 6,
    "analyses": [
      "query_fanout_coverage",
      "citation_volatility",
      "crawler_policy_divergence",
      "render_retrieval_gap",
      "entity_identity_drift",
      "agent_action_friction"
    ],
    "policy": [
      "No proprietary model weights, embeddings, rerankers, hidden prompts or private search-engine systems are claimed or inferred.",
      "Black-box risk means externally observable uncertainty or an evidence gap, not secret platform access.",
      "NOT_MEASURED and REQUIRES_CONTEXT remain excluded from all canonical scores.",
      "The 18-engine overall score and the 13 canonical Intelligence Audit analyses remain unchanged by this layer."
    ],
    "sourceIds": [
      "GOOGLE-GENAI-PERFORMANCE-2026",
      "GOOGLE-COMMON-CRAWLERS",
      "OPENAI-PUBLISHERS",
      "ANTHROPIC-CRAWLERS",
      "PERPLEXITY-ROBOTS",
      "GOOGLE-STRUCTURED-DATA",
      "WCAG22",
      "OPENAPI31"
    ],
    "version": "2.0.0",
    "transformerReverseEngineeringBoundary": "Observable retrieval, citation, rendering, entity and action behavior is tested as a black box. Proprietary transformer weights, embeddings, rerankers, hidden prompts and private indexes are neither accessed nor inferred as fact.",
    "observationContracts": [
      {
        "key": "query_fanout_coverage",
        "title": "Query Fan-Out Coverage",
        "hypothesis": "A single user question may fan out into multiple related retrieval intents, so landing-page coverage can fail even when the head query appears covered.",
        "observable": "Connected Search Console generative-AI performance dimensions and/or an authorized query-observation dataset mapped to canonical landing URLs.",
        "requiredInputs": [
          "authorized query observation data",
          "canonical landing URL inventory",
          "locale",
          "observation window"
        ],
        "metrics": [
          "covered intent clusters",
          "uncovered intent clusters",
          "landing-page overlap",
          "country/device divergence when supplied"
        ],
        "falsifier": "If repeated authorized observations show no material intent/landing divergence, the risk must remain low or NOT_MEASURED rather than being inferred from page copy.",
        "decisionRule": "No score is emitted from public HTML alone. Compare only like-for-like windows; provider-side hidden query branches are never reconstructed as fact.",
        "minimumEvidence": "At least two comparable observation windows before a trend claim; otherwise report a point-in-time observation only.",
        "temporalSensitivity": "HIGH",
        "sourceIds": [
          "GOOGLE-AI-OPTIMIZATION",
          "GOOGLE-GENAI-PERFORMANCE-2026"
        ],
        "lenses": [
          "SEO",
          "GEO",
          "AEO",
          "RAG"
        ],
        "failureModes": [
          "query taxonomy drift",
          "landing URL canonical changes",
          "insufficient connected data"
        ],
        "mitigations": [
          "version the query taxonomy",
          "segment before/after canonical changes",
          "preserve NOT_MEASURED when coverage data is incomplete"
        ]
      },
      {
        "key": "citation_volatility",
        "title": "Citation & Recommendation Volatility",
        "hypothesis": "Brand mentions, citations and recommendation position can vary by provider, prompt, locale and time even when the website is unchanged.",
        "observable": "Provider-backed response observations collected with fixed prompt IDs, locale, timestamp, surface and citation extraction.",
        "requiredInputs": [
          "configured provider surfaces",
          "versioned prompt IDs",
          "locale",
          "timestamps",
          "citation URLs"
        ],
        "metrics": [
          "mention rate",
          "citation rate",
          "recommendation rate",
          "share of answer",
          "citation-source churn"
        ],
        "falsifier": "If repeated fixed-protocol observations are stable within the measured window, the engine must not manufacture volatility from a single crawl.",
        "decisionRule": "Compare only identical protocol segments. A provider/model/surface change starts a new segment and invalidates direct before/after attribution.",
        "minimumEvidence": "Repeated observations across at least two comparable windows; a single response is never a trend.",
        "temporalSensitivity": "CRITICAL",
        "sourceIds": [
          "OPENAI-RESPONSES-WEBSEARCH",
          "PERPLEXITY-SONAR",
          "GEMINI-GOOGLE-SEARCH"
        ],
        "lenses": [
          "GEO",
          "AEO",
          "LLMO",
          "E-E-A-T"
        ],
        "failureModes": [
          "provider model update",
          "search grounding not triggered",
          "prompt drift",
          "citation parser ambiguity"
        ],
        "mitigations": [
          "segment by provider/model/surface",
          "record search-used/search-not-used state",
          "hash versioned prompts",
          "retain raw citation receipts"
        ]
      },
      {
        "key": "crawler_policy_divergence",
        "title": "Crawler Purpose & Policy Divergence",
        "hypothesis": "Search discovery, user-request retrieval and training crawlers can legitimately have different access policies; accidental cross-purpose blocking is the risk.",
        "observable": "Effective robots.txt policy and reachable HTTP behavior for identified crawler user agents.",
        "requiredInputs": [
          "robots.txt",
          "canonical URL",
          "crawler identity registry",
          "effective HTTP response"
        ],
        "metrics": [
          "allow/block by crawler",
          "purpose-class divergence",
          "missing explicit observation"
        ],
        "falsifier": "Purpose-specific divergence that matches the publisher's explicit policy is not a defect.",
        "decisionRule": "Flag only unexplained or contradictory effective policy. Never equate training access with search visibility.",
        "minimumEvidence": "One current effective-policy observation per crawler identity plus current first-party vendor guidance.",
        "temporalSensitivity": "HIGH",
        "sourceIds": [
          "OPENAI-PUBLISHERS",
          "GOOGLE-COMMON-CRAWLERS",
          "ANTHROPIC-CRAWLERS",
          "PERPLEXITY-ROBOTS"
        ],
        "lenses": [
          "SEO",
          "GEO",
          "LLMO"
        ],
        "failureModes": [
          "stale crawler names",
          "CDN/WAF user-agent blocking",
          "robots redirect/error"
        ],
        "mitigations": [
          "source-freshness gate",
          "probe effective HTTP path",
          "separate robots parsing from CDN/WAF denial"
        ]
      },
      {
        "key": "render_retrieval_gap",
        "title": "Render-to-Retrieval Gap",
        "hypothesis": "Critical text, links or structured evidence can exist only after client rendering, creating a retrieval gap for systems that consume raw or partially rendered HTML.",
        "observable": "Paired raw-response and controlled browser-render captures of the same canonical URL and timestamp window.",
        "requiredInputs": [
          "raw HTML snapshot",
          "rendered DOM snapshot",
          "canonical URL",
          "capture timestamps"
        ],
        "metrics": [
          "critical-text parity",
          "canonical parity",
          "structured-data parity",
          "primary-link parity",
          "content hash delta"
        ],
        "falsifier": "If critical content and machine-readable identity are equivalent across paired captures, the gap is not established.",
        "decisionRule": "Public raw-HTML scanning alone cannot pass or fail render parity. Missing browser evidence remains NOT_MEASURED.",
        "minimumEvidence": "Paired captures of the same URL under a controlled browser profile.",
        "temporalSensitivity": "MEDIUM",
        "sourceIds": [
          "GOOGLE-AI-OPTIMIZATION",
          "GOOGLE-DEVELOPER-SEO"
        ],
        "lenses": [
          "SEO",
          "GEO",
          "AEO",
          "RAG",
          "AAO"
        ],
        "failureModes": [
          "hydration timeout",
          "consent wall divergence",
          "geo/device conditional rendering",
          "capture race"
        ],
        "mitigations": [
          "bounded render timeout with explicit failure",
          "record consent state",
          "segment by device/locale",
          "capture raw and rendered artifacts under one correlation ID"
        ]
      },
      {
        "key": "entity_identity_drift",
        "title": "Entity Identity Drift",
        "hypothesis": "Conflicting names, URLs, prices, organization identity or product facts across owned machine-readable surfaces can increase ambiguity even without access to any private knowledge graph.",
        "observable": "Visible page facts, JSON-LD, canonical metadata and verified owned profiles supplied by the customer.",
        "requiredInputs": [
          "visible identity facts",
          "JSON-LD graph",
          "canonical URLs",
          "authorized owned-profile facts when available"
        ],
        "metrics": [
          "name consistency",
          "URL consistency",
          "offer/price consistency",
          "organization/product @id consistency",
          "conflict count"
        ],
        "falsifier": "If owned public identity facts are internally consistent, the engine must not demand Wikidata, a Google MID or any hidden identifier.",
        "decisionRule": "Score only public consistency evidence; cross-domain consensus remains REQUIRES_CONTEXT unless authoritative external evidence is supplied.",
        "minimumEvidence": "At least one visible fact source and one machine-readable representation for any claimed conflict.",
        "temporalSensitivity": "MEDIUM",
        "sourceIds": [
          "GOOGLE-STRUCTURED-DATA",
          "GOOGLE-ORGANIZATION-SCHEMA"
        ],
        "lenses": [
          "SEO",
          "GEO",
          "LLMO",
          "E-E-A-T"
        ],
        "failureModes": [
          "stale offer schema",
          "duplicate @id nodes",
          "www/apex inconsistency",
          "localized identity mismatch"
        ],
        "mitigations": [
          "bind visible and structured facts",
          "canonicalize @id graph",
          "normalize host policy",
          "validate locale-specific facts independently"
        ]
      },
      {
        "key": "agent_action_friction",
        "title": "Agent Action Friction",
        "hypothesis": "A site may be discoverable yet difficult for browser agents or assistive automation to act on because forms, controls, navigation or machine interfaces are ambiguous.",
        "observable": "Public accessibility, form labeling, action URLs, security boundaries, conversion path continuity and documented machine interfaces.",
        "requiredInputs": [
          "public action path",
          "form/control semantics",
          "HTTP status chain",
          "documented API/tool surface when intentionally published"
        ],
        "metrics": [
          "labeled-control coverage",
          "action-path continuity",
          "redirect/error count",
          "machine-interface discoverability",
          "auth-wall boundary"
        ],
        "falsifier": "If the target action is accessible, labeled and deterministic in the measured public flow, the engine must not claim autonomous-agent failure without an agent execution trace.",
        "decisionRule": "Website-side friction can be evaluated; autonomous purchase or task completion in an external agent is never guaranteed.",
        "minimumEvidence": "Measured public action path; external-agent completion requires a separate authorized execution trace.",
        "temporalSensitivity": "MEDIUM",
        "sourceIds": [
          "WCAG22",
          "OPENAPI31",
          "GOOGLE-AI-OPTIMIZATION"
        ],
        "lenses": [
          "AAO",
          "AEO",
          "E-E-A-T"
        ],
        "failureModes": [
          "unlabeled controls",
          "client-only navigation",
          "non-deterministic redirects",
          "undocumented auth requirement"
        ],
        "mitigations": [
          "programmatic labels",
          "crawlable/actionable links",
          "bounded redirect policy",
          "explicit auth and permission boundary"
        ]
      }
    ],
    "orchestrationContract": {
      "model": "N8N_INSPIRED_EVENT_DRIVEN_OBSERVATION_CONTRACT",
      "stages": [
        "INGEST_EVIDENCE",
        "VALIDATE_AND_NORMALIZE",
        "OBSERVE_WITH_FIXED_PROTOCOL",
        "COMPARE_LIKE_FOR_LIKE",
        "DECIDE_WITH_EPISTEMIC_GATE",
        "PERSIST_RECEIPT_OR_DLQ"
      ],
      "correlationKey": "scanId + domain + riskKey + protocolVersion",
      "idempotencyKey": "sha256(domain|riskKey|windowStart|protocolVersion)",
      "retryPolicy": {
        "classification": "IMPLEMENTATION_DEFAULT_NOT_VENDOR_CLAIM",
        "retryableHttp": [
          408,
          429,
          500,
          502,
          503,
          504
        ],
        "maxRetries": 2,
        "backoffMs": [
          1000,
          4000
        ],
        "nonRetryable": [
          "invalid target",
          "authorization failure",
          "policy denial",
          "schema/contract violation"
        ]
      },
      "dlq": {
        "on": [
          "retry exhaustion",
          "malformed provider receipt",
          "protocol mismatch",
          "stale or missing required evidence"
        ],
        "effectOnScore": "NONE",
        "requiredReceipt": [
          "correlationKey",
          "riskKey",
          "failureClass",
          "attemptCount",
          "timestamp",
          "evidenceRefs"
        ]
      },
      "temporalIntegrity": [
        "Provider/model/surface changes start a new comparison segment.",
        "A missing observation cannot be converted into PASS or FAIL.",
        "Removing evidence cannot increase confidence.",
        "A point-in-time result cannot be labeled a trend."
      ],
      "security": [
        "Provider credentials stay server-side and are never serialized into scan reports.",
        "Public scanning remains read-only and SSRF fail-closed.",
        "Tool or provider output is treated as untrusted data until schema-validated."
      ]
    },
    "decisionStates": [
      "PASS",
      "WARN",
      "FAIL",
      "NOT_MEASURED",
      "REQUIRES_CONTEXT"
    ],
    "confidenceMonotonicity": "Confidence must not increase when evidence is removed or becomes stale.",
    "comparisonRule": "Only like-for-like protocol segments may be compared as before/after evidence.",
    "publicContract": "https://htmlandhtml.com/blackbox-observation-contract.json"
  },
  "publicPositioning": {
    "category": "AI Search Technical Diagnostic Platform",
    "primarySearchIntentsTR": [
      "yapay zeka seo analizi",
      "chatgpt görünürlük",
      "yapay zeka arama görünürlüğü",
      "llms.txt validator",
      "ai crawler checker"
    ],
    "primarySearchIntentsEN": [
      "AI SEO audit",
      "ChatGPT visibility",
      "AI search visibility",
      "llms.txt validator",
      "AI crawler checker"
    ],
    "guaranteeBoundary": "No ranking, citation, recommendation, traffic or revenue guarantee."
  }
}
