File viewer

draft_12082026_0223.json

/app/data/llm/draft/bupa/draft_12082026_0223.json

{
  "summary_points": [
    "AI bots have open access. Live checks show 0% block rate and 42–84 ms response times for all major crawlers, eliminating blocking as a cause.",
    "Template signal failure is the root cause. ~4,507 empty H1s and ~4,503 missing meta descriptions mean deep pages lack extractable structure, forcing AI models to cite only the homepage.",
    "Competitor Bowtie outperforms structurally. Bowtie earns 48 responses from 47 cited deep pages versus Bupa's 29 responses from 26 pages (homepage 34%), capturing the highest-volume commercial query.",
    "The fix is on-site and fast. The primary need is a template deployment to restore H1 and meta description signals, not a multi-quarter authority campaign."
  ],
  "method_notes": "This analysis combines live-crawl checks (5,191 URLs, UA-based from a single datacenter IP), a deep crawl (5,191 URLs seeded mid-architecture), and Brand Radar data (AI query demand, citation counts, share of voice). The gap closed: eliminated blocking and content volume as explanations, isolating template signal failure as the mechanism. Correction to prior work: Brand Radar framed the citation gap as 'a content structure and citation authority problem, not a technical SEO problem'; the crawl shows it is a template-level technical problem—empty H1s and missing meta descriptions are structural failures.",
  "featured_finding": {
    "headline": "Bupa's AI invisibility is a self-inflicted, on-site problem.",
    "narrative": "Every data source agrees – the issue is not access, content volume, or authority. Live checks prove every major AI crawler (GPTBot, ClaudeBot, PerplexityBot, OAI-SearchBot, Google-Extended, Bytespider) reaches bupa.com.hk unimpeded: 0% block rate, sub-90 ms response times. Brand Radar shows that despite open access, Bupa owns only 26 of 389 cited pages (6.7% own-citation share) in its own AI query set. Its most-cited page is the homepage (10 of 29 responses), while no product detail, claims, or network hospital page is cited once. Meanwhile Bowtie, a direct competitor, earns 48 responses from 47 deep pages and answers the 2,100-volume Traditional Chinese query '香港自愿医保哪家好?' without Bupa being referenced. The crawl reveals why: ~4,507 URLs crawled had empty H1s, and ~4,503 had missing meta descriptions. This is a near-universal template failure that strips deep pages of the heading and descriptive signals AI models extract when deciding what to cite. The mechanism is eliminative: blocking is ruled out, content absence is ruled out (1,216 /tc/ URLs exist and are crawlable), so the citation gap is consistent with pages that present no extractable structure. The confirming test is simple – cross-walk Bupa's 26 cited pages and Bowtie's 47 cited pages against crawl H1 presence to verify the structural difference.",
    "proof_points": [
      "Live checks: 0% block rate, 42–84 ms response times for all major AI crawlers across all paths.",
      "Brand Radar: Bupa owns 26 of 389 cited pages (6.7% own-citation share); homepage accounts for 10 of 29 responses; zero product/claims/network pages cited.",
      "Crawl: ~4,507 of 5,191 crawled URLs have empty H1s; ~4,503 lack meta descriptions.",
      "Together this means the mechanism is template signal failure – not blocking, not content volume, not authority."
    ],
    "why_it_wins": "This is the most important finding because it sits on the highest-value demand (12,420 monthly AI impressions, 70.4% via Google AI Overviews, led by a 2,100-volume commercial query), it is losing that demand to a named direct competitor (Bowtie), the fix is a template deployment rather than a multi-quarter authority campaign, and every alternative explanation (blocking, content absence, authority) has already been eliminated by the data in hand. Nothing else in the analysis – redirects, crawl budget, meta description CTR – comes close to this combination of value at stake, clarity of cause, and speed of fix."
  },
  "key_findings": [
    {
      "finding": "AI citation gap is driven by template signal failure, not by access or content volume.",
      "evidence": "Live checks: all AI bots return 200, 42–84ms. Brand Radar: Bupa cited on only 29 of 389 pages; homepage 34% of own citations; no deep pages cited. Crawl: ~4,507 URLs with empty H1s, ~4,503 with missing meta descriptions across 5,191 crawled URLs.",
      "implication": "AI models cannot extract structured information from Bupa's deep pages, so they default to citing the homepage or, more often, competitors and aggregators with clearer signals."
    },
    {
      "finding": "Bowtie's structural advantage gives it dominant citation share on high-volume queries.",
      "evidence": "Bowtie earns 48 responses from 47 cited deep pages vs. Bupa's 29 responses from 26 pages. The 2,100-volume Traditional Chinese query '香港自愿医保哪家好?' is answered by Bowtie, 10life, hkvhis, and moneyhk101 – not Bupa.",
      "implication": "Competitor pages are better structured for extraction, allowing them to capture demand Bupa should own given its brand strength and /tc/ content library."
    },
    {
      "finding": "Crawl budget is diluted by redirects and junk paths, slowing discovery of citable content.",
      "evidence": "23% of crawled URLs are redirects; /files/, /_Incapsula_Resource/, and /PDF/ together account for ~30% of crawled URLs. Only 5,191 of ~34K URLs were reached.",
      "implication": "Every bot – Googlebot or AI crawler – spends fetch budget on non-content pages, delaying indexing and refresh of the /tc/ and /en/ editorial pages that need to be cited."
    },
    {
      "finding": "Brand Radar's share of voice is a measurement artefact with no competitive signal.",
      "evidence": "100% share of voice with a flat 365-day trendline indicates only Bupa is tracked. Real competitive metrics are own-citation share (6.7%) and Bowtie's 48 responses vs Bupa's 29.",
      "implication": "The current Brand Radar configuration cannot be used as a competitive benchmark; competitor tracking must be added for meaningful SOV data."
    },
    {
      "finding": "Language mismatch: high-volume Traditional Chinese queries are lost despite large /tc/ content library.",
      "evidence": "Brand Radar shows highest-volume AI queries are Traditional Chinese (e.g., 2,100 vol). Crawl shows 1,216 /tc/ URLs exist. Yet no Bupa page answers those queries; competitors and aggregators do.",
      "implication": "The /tc/ content may be structurally invisible to AI models due to the same H1/meta failure, making language targeting useless without template fix."
    }
  ],
  "risk_areas": [
    {
      "issue": "Near-universal empty H1s and missing meta descriptions on crawled editorial pages.",
      "priority": "P0",
      "effort": "HIGH",
      "impact": "HIGH",
      "detail": "~4,507 URLs lack H1, ~4,503 lack meta description across 5,191 crawled, including the entire /tc/, /en/, /zh/ directories. This directly prevents AI models from extracting page identity and relevance."
    },
    {
      "issue": "robots.txt content unknown – could be blocking compliant AI bots despite HTTP 200s.",
      "priority": "P1",
      "effort": "LOW",
      "impact": "HIGH",
      "detail": "Live checks only verify HTTP-level access; GPTBot and ClaudeBot obey robots.txt. If robots.txt has Disallow directives for AI agents, the entire GEO narrative is compromised."
    },
    {
      "issue": "Crawl budget wasted on redirects and non-content paths.",
      "priority": "P1",
      "effort": "MEDIUM",
      "impact": "MEDIUM",
      "detail": "23% redirects, and /files/, /_Incapsula_Resource/, /PDF/ account for ~30% of crawled URLs. This reduces the effective crawl rate for /tc/ and /en/ citable pages."
    },
    {
      "issue": "No cross-walk between cited pages and crawl H1 data to confirm mechanism.",
      "priority": "P2",
      "effort": "LOW",
      "impact": "MEDIUM",
      "detail": "Both datasets exist but have not been joined to verify whether the 26 cited pages are the rare ones with intact H1s. This would convert 'consistent with' into proven cause."
    }
  ],
  "bot_management": {
    "observed_facts": [
      "Live checks returned 200s for all tested AI bots (GPTBot, ClaudeBot, PerplexityBot, OAI-SearchBot, Google-Extended, Bytespider) on all paths, with response times 42–84 ms.",
      "PerplexityBot is explicitly allowed (200).",
      "A datacenter IP (non-bot) returned 403 for some paths, but this does not indicate crawler blocking."
    ],
    "seo_risks": [
      "No SEO risk from bot blocking identified from live checks, because all AI crawlers got 200s."
    ],
    "geo_hypothesis": {
      "claim": "Bot-blocking is not a factor in Bupa's AI citation gap, but if robots.txt restricts compliant AI agents it would worsen the gap.",
      "supporting": [
        "Live checks show all major AI crawlers get 200 and fast response times.",
        "Brand Radar shows only 29 citations from 389 pages; the failure is in extraction, not access."
      ],
      "prevents_calling_fact": [
        "We have not fetched robots.txt to verify per-agent rules.",
        "Live checks are UA-based from a single datacenter IP and cannot simulate real AI crawler behaviour from distributed IPs."
      ],
      "definitive_check": "Fetch robots.txt and check for Disallow directives targeting GPTBot, ClaudeBot, OAI-SearchBot, Google-Extended, Bytespider.",
      "recommendations": [
        "Immediately audit robots.txt for any restrictive directives against AI agents.",
        "If restrictive directives exist, remove them to ensure full open access for AI crawlers.",
        "After audit, confirm via server logs that AI bots are actually fetching content pages."
      ],
      "method_caveat": "This test is UA-based from a single datacenter IP and is confounded; it does not verify robots.txt compliance nor production-scale bot behaviour."
    },
    "recommendations": [
      "Audit robots.txt for AI bot disallow directives.",
      "If none exist, no further action on bot management.",
      "If restrictive, remove and monitor AI citations via Brand Radar after a refresh period."
    ],
    "method_caveat": "Live checks only prove HTTP-level access for the tested UAs and IP; real AI bot behaviour could differ due to robots.txt or rate limits."
  },
  "narrative_arc": [
    "We identified the root cause of Bupa's AI invisibility – it's on-site and structural, not off-site or authority-based.",
    "Three independent data sources converge: access is open, content exists, but templates strip every page of extractable signals.",
    "The fix is faster than you think – a template deployment for H1s and meta descriptions will unlock citation of deep product, claims, and network pages.",
    "We can verify the precise mechanism in week 1 by cross-walking cited pages with crawl data – a task your team can do immediately after granting GSC access."
  ],
  "recommendations": [
    {
      "action": "Audit robots.txt to confirm no restrictive directives against AI agents.",
      "rationale": "Gates the entire GEO narrative; if robots.txt blocks GPTBot or ClaudeBot, even template fixes won't matter.",
      "validation": "Fetch robots.txt, check for Disallow on GPTBot, ClaudeBot, OAI-SearchBot, Google-Extended, Bytespider."
    },
    {
      "action": "Cross-walk the 26 cited Bupa pages and 47 Bowtie cited pages against crawl H1/meta data to confirm structural difference.",
      "rationale": "Converts 'consistent with' into proof of mechanism; client team can do this immediately with existing datasets.",
      "validation": "Compare H1 presence percentage in cited vs uncited pages; document gap."
    },
    {
      "action": "Clean crawl budget by removing 23% redirects and excluding /files/, /_Incapsula_Resource/, /PDF/ from crawl paths.",
      "rationale": "Frees up fetch budget for citable /tc/ and /en/ content, speeding up indexing and refresh.",
      "validation": "Re-crawl after cleanup; confirm increase in deep-page indexation in GSC."
    },
    {
      "action": "Implement a template fix to restore H1 and meta description on all editorial pages in /tc/, /en/, /zh/.",
      "rationale": "~4,507 URLs currently lack H1s and ~4,503 lack meta descriptions; these are the primary signals AI models use for citation.",
      "validation": "Post-fix crawl to confirm 100% H1 and meta description coverage on indexable pages."
    },
    {
      "action": "Configure Brand Radar to track competitors (Bowtie, AXA, etc.) for meaningful share of voice comparison.",
      "rationale": "Current 100% SOV is an artefact; real competitive tracking is needed to measure impact.",
      "validation": "Confirm competitor pages appear in Brand Radar query sets."
    }
  ],
  "hypotheses": [
    {
      "claim": "Template signal failure is the mechanism causing the AI citation gap.",
      "status": "suggestive",
      "how_to_verify": "Cross-walk cited pages with crawl H1 presence: if cited pages are the minority with intact H1s, hypothesis is confirmed."
    },
    {
      "claim": "robots.txt may block compliant AI bots despite HTTP 200s.",
      "status": "unverified",
      "how_to_verify": "Fetch robots.txt and check for Disallow directives for GPTBot, ClaudeBot, OAI-SearchBot, Google-Extended, Bytespider."
    },
    {
      "claim": "Bowtie's cited pages are structurally better with H1s and meta descriptions.",
      "status": "suggestive",
      "how_to_verify": "Crawl Bowtie's 47 cited pages and compare H1/meta coverage vs Bupa's cited and uncited pages."
    }
  ]
}