{
  "study": "GitLab public root-page access checks",
  "date": "2026-09-06",
  "environment": "Ordinary-IP local research runtime; authenticated crawler IPs, egress IP and location not independently recorded.",
  "limits": "HTTP header observations; not verified real crawler access, page content, indexing, AI citations or client outcomes.",
  "rounds": [
    {
      "round": "initial",
      "scannedAt": "2026-09-06T17:41:41.739Z",
      "requestedUrl": "https://about.gitlab.com/",
      "robots": {
        "url": "https://about.gitlab.com/robots.txt",
        "finalUrl": "https://about.gitlab.com/robots.txt",
        "httpStatus": 200,
        "state": "available",
        "userAgent": "DeepOceanCrawlerCheck/1.0 (+https://deepoceanstudio.com/scan/)",
        "note": "Policy fetched once with the displayed checker user-agent. Rules below apply to the submitted URL; responses can vary by user-agent or IP."
      },
      "crawlers": [
        {
          "id": "oai-searchbot",
          "name": "OAI-SearchBot",
          "provider": "OpenAI",
          "purpose": "ChatGPT search",
          "category": "search",
          "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:41.236Z",
            "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "googlebot",
          "name": "Googlebot",
          "provider": "Google",
          "purpose": "Google Search, including AI features",
          "category": "search",
          "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:41.252Z",
            "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "perplexitybot",
          "name": "PerplexityBot",
          "provider": "Perplexity",
          "purpose": "Perplexity search",
          "category": "search",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:40.918Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claude-searchbot",
          "name": "Claude-SearchBot",
          "provider": "Anthropic",
          "purpose": "Claude search",
          "category": "search",
          "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:41.389Z",
            "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "gptbot",
          "name": "GPTBot",
          "provider": "OpenAI",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:41.709Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claudebot",
          "name": "ClaudeBot",
          "provider": "Anthropic",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:41.739Z",
            "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "google-extended",
          "name": "Google-Extended",
          "provider": "Google",
          "purpose": "Gemini training and grounding policy",
          "category": "control",
          "userAgent": null,
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "not_applicable",
            "status": null,
            "finalUrl": null,
            "checkedAt": "2026-09-06T17:41:41.389Z",
            "userAgent": null,
            "note": "Google-Extended is a robots policy token. It has no separate HTTP crawler user-agent to test."
          }
        }
      ],
      "scope": "Robots rules for the submitted URL and separate HTTP user-agent simulations from an ordinary IP. This checks access conditions; it does not establish real crawler visits, indexing, AI-answer presence or citations. Redirect destinations can have different robots policies. Server/CDN logs or official crawler IP verification are needed to confirm real crawler access."
    },
    {
      "round": "commercial-root repeat",
      "scannedAt": "2026-09-06T17:42:46.716Z",
      "requestedUrl": "https://about.gitlab.com/",
      "robots": {
        "url": "https://about.gitlab.com/robots.txt",
        "finalUrl": "https://about.gitlab.com/robots.txt",
        "httpStatus": 200,
        "state": "available",
        "userAgent": "DeepOceanCrawlerCheck/1.0 (+https://deepoceanstudio.com/scan/)",
        "note": "Policy fetched once with the displayed checker user-agent. Rules below apply to the submitted URL; responses can vary by user-agent or IP."
      },
      "crawlers": [
        {
          "id": "oai-searchbot",
          "name": "OAI-SearchBot",
          "provider": "OpenAI",
          "purpose": "ChatGPT search",
          "category": "search",
          "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:42:45.917Z",
            "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "googlebot",
          "name": "Googlebot",
          "provider": "Google",
          "purpose": "Google Search, including AI features",
          "category": "search",
          "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:42:46.257Z",
            "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "perplexitybot",
          "name": "PerplexityBot",
          "provider": "Perplexity",
          "purpose": "Perplexity search",
          "category": "search",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:42:46.240Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claude-searchbot",
          "name": "Claude-SearchBot",
          "provider": "Anthropic",
          "purpose": "Claude search",
          "category": "search",
          "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:42:46.394Z",
            "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "gptbot",
          "name": "GPTBot",
          "provider": "OpenAI",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:42:46.711Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claudebot",
          "name": "ClaudeBot",
          "provider": "Anthropic",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:42:46.716Z",
            "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "google-extended",
          "name": "Google-Extended",
          "provider": "Google",
          "purpose": "Gemini training and grounding policy",
          "category": "control",
          "userAgent": null,
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "not_applicable",
            "status": null,
            "finalUrl": null,
            "checkedAt": "2026-09-06T17:42:46.394Z",
            "userAgent": null,
            "note": "Google-Extended is a robots policy token. It has no separate HTTP crawler user-agent to test."
          }
        }
      ],
      "scope": "Robots rules for the submitted URL and separate HTTP user-agent simulations from an ordinary IP. This checks access conditions; it does not establish real crawler visits, indexing, AI-answer presence or citations. Redirect destinations can have different robots policies. Server/CDN logs or official crawler IP verification are needed to confirm real crawler access."
    },
    {
      "round": "prepublication verification",
      "scannedAt": "2026-09-06T17:47:45.914Z",
      "requestedUrl": "https://about.gitlab.com/",
      "robots": {
        "url": "https://about.gitlab.com/robots.txt",
        "finalUrl": "https://about.gitlab.com/robots.txt",
        "httpStatus": 200,
        "state": "available",
        "userAgent": "DeepOceanCrawlerCheck/1.0 (+https://deepoceanstudio.com/scan/)",
        "note": "Policy fetched once with the displayed checker user-agent. Rules below apply to the submitted URL; responses can vary by user-agent or IP."
      },
      "crawlers": [
        {
          "id": "oai-searchbot",
          "name": "OAI-SearchBot",
          "provider": "OpenAI",
          "purpose": "ChatGPT search",
          "category": "search",
          "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:45.118Z",
            "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "googlebot",
          "name": "Googlebot",
          "provider": "Google",
          "purpose": "Google Search, including AI features",
          "category": "search",
          "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:45.443Z",
            "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "perplexitybot",
          "name": "PerplexityBot",
          "provider": "Perplexity",
          "purpose": "Perplexity search",
          "category": "search",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:45.430Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claude-searchbot",
          "name": "Claude-SearchBot",
          "provider": "Anthropic",
          "purpose": "Claude search",
          "category": "search",
          "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:45.592Z",
            "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "gptbot",
          "name": "GPTBot",
          "provider": "OpenAI",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:45.909Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claudebot",
          "name": "ClaudeBot",
          "provider": "Anthropic",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "refused",
            "status": 403,
            "finalUrl": "https://about.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:45.914Z",
            "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "google-extended",
          "name": "Google-Extended",
          "provider": "Google",
          "purpose": "Gemini training and grounding policy",
          "category": "control",
          "userAgent": null,
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "not_applicable",
            "status": null,
            "finalUrl": null,
            "checkedAt": "2026-09-06T17:47:45.592Z",
            "userAgent": null,
            "note": "Google-Extended is a robots policy token. It has no separate HTTP crawler user-agent to test."
          }
        }
      ],
      "scope": "Robots rules for the submitted URL and separate HTTP user-agent simulations from an ordinary IP. This checks access conditions; it does not establish real crawler visits, indexing, AI-answer presence or citations. Redirect destinations can have different robots policies. Server/CDN logs or official crawler IP verification are needed to confirm real crawler access.",
      "robotsCaptures": [
        {
          "url": "https://about.gitlab.com/robots.txt",
          "checkedAt": "2026-09-06T17:47:44.950Z",
          "status": 200,
          "body": "# START nuxt-robots (indexable)\nUser-agent: *\nDisallow: /search/\nDisallow: /de-de/search/\nDisallow: /es/search/\nDisallow: /fr-fr/search/\nDisallow: /it-it/search/\nDisallow: /ja-jp/search/\nDisallow: /ko-kr/search/\nDisallow: /pt-br/search/\nDisallow: /api/\n\nSitemap: https://about.gitlab.com/sitemap.xml\nSitemap: https://about.gitlab.com/sitemap_index.xml\n# END nuxt-robots"
        }
      ]
    },
    {
      "round": "initial",
      "scannedAt": "2026-09-06T17:41:43.640Z",
      "requestedUrl": "https://docs.gitlab.com/",
      "robots": {
        "url": "https://docs.gitlab.com/robots.txt",
        "finalUrl": "https://docs.gitlab.com/robots.txt",
        "httpStatus": 200,
        "state": "available",
        "userAgent": "DeepOceanCrawlerCheck/1.0 (+https://deepoceanstudio.com/scan/)",
        "note": "Policy fetched once with the displayed checker user-agent. Rules below apply to the submitted URL; responses can vary by user-agent or IP."
      },
      "crawlers": [
        {
          "id": "oai-searchbot",
          "name": "OAI-SearchBot",
          "provider": "OpenAI",
          "purpose": "ChatGPT search",
          "category": "search",
          "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:42.996Z",
            "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "googlebot",
          "name": "Googlebot",
          "provider": "Google",
          "purpose": "Google Search, including AI features",
          "category": "search",
          "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:43.160Z",
            "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "perplexitybot",
          "name": "PerplexityBot",
          "provider": "Perplexity",
          "purpose": "Perplexity search",
          "category": "search",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:43.163Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claude-searchbot",
          "name": "Claude-SearchBot",
          "provider": "Anthropic",
          "purpose": "Claude search",
          "category": "search",
          "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:43.479Z",
            "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "gptbot",
          "name": "GPTBot",
          "provider": "OpenAI",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:43.638Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claudebot",
          "name": "ClaudeBot",
          "provider": "Anthropic",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:43.640Z",
            "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "google-extended",
          "name": "Google-Extended",
          "provider": "Google",
          "purpose": "Gemini training and grounding policy",
          "category": "control",
          "userAgent": null,
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "not_applicable",
            "status": null,
            "finalUrl": null,
            "checkedAt": "2026-09-06T17:41:43.479Z",
            "userAgent": null,
            "note": "Google-Extended is a robots policy token. It has no separate HTTP crawler user-agent to test."
          }
        }
      ],
      "scope": "Robots rules for the submitted URL and separate HTTP user-agent simulations from an ordinary IP. This checks access conditions; it does not establish real crawler visits, indexing, AI-answer presence or citations. Redirect destinations can have different robots policies. Server/CDN logs or official crawler IP verification are needed to confirm real crawler access."
    },
    {
      "round": "prepublication verification",
      "scannedAt": "2026-09-06T17:47:47.884Z",
      "requestedUrl": "https://docs.gitlab.com/",
      "robots": {
        "url": "https://docs.gitlab.com/robots.txt",
        "finalUrl": "https://docs.gitlab.com/robots.txt",
        "httpStatus": 200,
        "state": "available",
        "userAgent": "DeepOceanCrawlerCheck/1.0 (+https://deepoceanstudio.com/scan/)",
        "note": "Policy fetched once with the displayed checker user-agent. Rules below apply to the submitted URL; responses can vary by user-agent or IP."
      },
      "crawlers": [
        {
          "id": "oai-searchbot",
          "name": "OAI-SearchBot",
          "provider": "OpenAI",
          "purpose": "ChatGPT search",
          "category": "search",
          "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:47.083Z",
            "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "googlebot",
          "name": "Googlebot",
          "provider": "Google",
          "purpose": "Google Search, including AI features",
          "category": "search",
          "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:47.395Z",
            "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "perplexitybot",
          "name": "PerplexityBot",
          "provider": "Perplexity",
          "purpose": "Perplexity search",
          "category": "search",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:47.399Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claude-searchbot",
          "name": "Claude-SearchBot",
          "provider": "Anthropic",
          "purpose": "Claude search",
          "category": "search",
          "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:47.554Z",
            "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "gptbot",
          "name": "GPTBot",
          "provider": "OpenAI",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:47.878Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claudebot",
          "name": "ClaudeBot",
          "provider": "Anthropic",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://docs.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:47.884Z",
            "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "google-extended",
          "name": "Google-Extended",
          "provider": "Google",
          "purpose": "Gemini training and grounding policy",
          "category": "control",
          "userAgent": null,
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "not_applicable",
            "status": null,
            "finalUrl": null,
            "checkedAt": "2026-09-06T17:47:47.554Z",
            "userAgent": null,
            "note": "Google-Extended is a robots policy token. It has no separate HTTP crawler user-agent to test."
          }
        }
      ],
      "scope": "Robots rules for the submitted URL and separate HTTP user-agent simulations from an ordinary IP. This checks access conditions; it does not establish real crawler visits, indexing, AI-answer presence or citations. Redirect destinations can have different robots policies. Server/CDN logs or official crawler IP verification are needed to confirm real crawler access.",
      "robotsCaptures": [
        {
          "url": "https://docs.gitlab.com/robots.txt",
          "checkedAt": "2026-09-06T17:47:46.909Z",
          "status": 200,
          "body": "User-agent: *\n# Block crawlers from review apps\nDisallow: /review-mr\nDisallow: /upstream-review-mr\n# Block crawlers from released versions\nDisallow: /1\nDisallow: /2\nDisallow: /3\nSitemap: https://docs.gitlab.com/sitemap.xml"
        }
      ]
    },
    {
      "round": "initial",
      "scannedAt": "2026-09-06T17:41:45.979Z",
      "requestedUrl": "https://handbook.gitlab.com/",
      "robots": {
        "url": "https://handbook.gitlab.com/robots.txt",
        "finalUrl": "https://handbook.gitlab.com/robots.txt",
        "httpStatus": 200,
        "state": "available",
        "userAgent": "DeepOceanCrawlerCheck/1.0 (+https://deepoceanstudio.com/scan/)",
        "note": "Policy fetched once with the displayed checker user-agent. Rules below apply to the submitted URL; responses can vary by user-agent or IP."
      },
      "crawlers": [
        {
          "id": "oai-searchbot",
          "name": "OAI-SearchBot",
          "provider": "OpenAI",
          "purpose": "ChatGPT search",
          "category": "search",
          "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:45.065Z",
            "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "googlebot",
          "name": "Googlebot",
          "provider": "Google",
          "purpose": "Google Search, including AI features",
          "category": "search",
          "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:45.419Z",
            "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "perplexitybot",
          "name": "PerplexityBot",
          "provider": "Perplexity",
          "purpose": "Perplexity search",
          "category": "search",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:45.415Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claude-searchbot",
          "name": "Claude-SearchBot",
          "provider": "Anthropic",
          "purpose": "Claude search",
          "category": "search",
          "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:45.609Z",
            "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "gptbot",
          "name": "GPTBot",
          "provider": "OpenAI",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:45.979Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claudebot",
          "name": "ClaudeBot",
          "provider": "Anthropic",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:41:45.965Z",
            "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "google-extended",
          "name": "Google-Extended",
          "provider": "Google",
          "purpose": "Gemini training and grounding policy",
          "category": "control",
          "userAgent": null,
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "not_applicable",
            "status": null,
            "finalUrl": null,
            "checkedAt": "2026-09-06T17:41:45.609Z",
            "userAgent": null,
            "note": "Google-Extended is a robots policy token. It has no separate HTTP crawler user-agent to test."
          }
        }
      ],
      "scope": "Robots rules for the submitted URL and separate HTTP user-agent simulations from an ordinary IP. This checks access conditions; it does not establish real crawler visits, indexing, AI-answer presence or citations. Redirect destinations can have different robots policies. Server/CDN logs or official crawler IP verification are needed to confirm real crawler access."
    },
    {
      "round": "prepublication verification",
      "scannedAt": "2026-09-06T17:47:50.178Z",
      "requestedUrl": "https://handbook.gitlab.com/",
      "robots": {
        "url": "https://handbook.gitlab.com/robots.txt",
        "finalUrl": "https://handbook.gitlab.com/robots.txt",
        "httpStatus": 200,
        "state": "available",
        "userAgent": "DeepOceanCrawlerCheck/1.0 (+https://deepoceanstudio.com/scan/)",
        "note": "Policy fetched once with the displayed checker user-agent. Rules below apply to the submitted URL; responses can vary by user-agent or IP."
      },
      "crawlers": [
        {
          "id": "oai-searchbot",
          "name": "OAI-SearchBot",
          "provider": "OpenAI",
          "purpose": "ChatGPT search",
          "category": "search",
          "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:49.270Z",
            "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "googlebot",
          "name": "Googlebot",
          "provider": "Google",
          "purpose": "Google Search, including AI features",
          "category": "search",
          "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:49.602Z",
            "userAgent": "Googlebot/2.1 (+http://www.google.com/bot.html)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "perplexitybot",
          "name": "PerplexityBot",
          "provider": "Perplexity",
          "purpose": "Perplexity search",
          "category": "search",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:49.610Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claude-searchbot",
          "name": "Claude-SearchBot",
          "provider": "Anthropic",
          "purpose": "Claude search",
          "category": "search",
          "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:49.824Z",
            "userAgent": "Claude-SearchBot/1.0; +https://www.anthropic.com/claude-searchbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "gptbot",
          "name": "GPTBot",
          "provider": "OpenAI",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:50.148Z",
            "userAgent": "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.4; +https://openai.com/gptbot",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "claudebot",
          "name": "ClaudeBot",
          "provider": "Anthropic",
          "purpose": "Model training",
          "category": "training",
          "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "reachable",
            "status": 200,
            "finalUrl": "https://handbook.gitlab.com/",
            "checkedAt": "2026-09-06T17:47:50.178Z",
            "userAgent": "Mozilla/5.0 (compatible; ClaudeBot/1.0; +claudebot@anthropic.com)",
            "note": "HTTP GET response from an ordinary IP using this user-agent string. The real crawler may receive a different response. Only response headers were fetched."
          }
        },
        {
          "id": "google-extended",
          "name": "Google-Extended",
          "provider": "Google",
          "purpose": "Gemini training and grounding policy",
          "category": "control",
          "userAgent": null,
          "robots": {
            "state": "allowed",
            "matchedRule": null,
            "matchedGroups": [
              "*"
            ],
            "note": "No matching disallow rule for the submitted path in the fetched policy."
          },
          "http": {
            "state": "not_applicable",
            "status": null,
            "finalUrl": null,
            "checkedAt": "2026-09-06T17:47:49.824Z",
            "userAgent": null,
            "note": "Google-Extended is a robots policy token. It has no separate HTTP crawler user-agent to test."
          }
        }
      ],
      "scope": "Robots rules for the submitted URL and separate HTTP user-agent simulations from an ordinary IP. This checks access conditions; it does not establish real crawler visits, indexing, AI-answer presence or citations. Redirect destinations can have different robots policies. Server/CDN logs or official crawler IP verification are needed to confirm real crawler access.",
      "robotsCaptures": [
        {
          "url": "https://handbook.gitlab.com/robots.txt",
          "checkedAt": "2026-09-06T17:47:49.062Z",
          "status": 200,
          "body": "User-agent: *\n"
        }
      ]
    }
  ],
  "browserControls": [
    {
      "url": "https://about.gitlab.com/",
      "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
      "status": 200,
      "finalUrl": "https://about.gitlab.com/",
      "checkedAt": "2026-09-06T17:47:46.392Z"
    },
    {
      "url": "https://docs.gitlab.com/",
      "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
      "status": 200,
      "finalUrl": "https://docs.gitlab.com/",
      "checkedAt": "2026-09-06T17:47:48.453Z"
    },
    {
      "url": "https://handbook.gitlab.com/",
      "userAgent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
      "status": 200,
      "finalUrl": "https://handbook.gitlab.com/",
      "checkedAt": "2026-09-06T17:47:50.754Z"
    }
  ]
}
