# ============================================================================ # robots.txt for cnglobalsourcing.com # Last updated: 2026-07-19 # ============================================================================ # ── All crawlers: allow everything except private/admin paths ──────────────── User-agent: * Allow: / Disallow: /admin/ Disallow: /private/ Disallow: /cgi-bin/ Disallow: /tmp/ # ── Sitemap references (all major search engines) ───────────────────────────── Sitemap: https://cnglobalsourcing.com/sitemap.xml # ── Google / Googlebot ──────────────────────────────────────────────────────── User-agent: Googlebot Allow: / Crawl-delay: 0 User-agent: Googlebot-Image Allow: / Crawl-delay: 0 User-agent: Googlebot-News Allow: / Crawl-delay: 0 User-agent: Googlebot-Video Allow: / Crawl-delay: 0 User-agent: Googlebot-Mobile Allow: / Crawl-delay: 0 User-agent: AdsBot-Google Allow: / Crawl-delay: 0 # ── Baidu / Baiduspider ────────────────────────────────────────────────────── User-agent: Baiduspider Allow: / Crawl-delay: 1 User-agent: Baiduspider-image Allow: / Crawl-delay: 1 User-agent: Baiduspider-video Allow: / Crawl-delay: 1 User-agent: Baiduspider-mobile Allow: / Crawl-delay: 1 # ── Bing / Bingbot ─────────────────────────────────────────────────────────── User-agent: Bingbot Allow: / Crawl-delay: 0 User-agent: BingPreview Allow: / Crawl-delay: 0 # ── Yandex / YandexBot ─────────────────────────────────────────────────────── User-agent: YandexBot Allow: / Crawl-delay: 1 User-agent: YandexImages Allow: / Crawl-delay: 1 User-agent: YandexVideo Allow: / Crawl-delay: 1 User-agent: YandexMobileBot Allow: / Crawl-delay: 1 User-agent: YandexMetrika Allow: / Crawl-delay: 1 # ── Naver / Yeti ─────────────────────────────────────────────────────────── User-agent: Yeti Allow: / Crawl-delay: 1 User-agent: NaverBot Allow: / Crawl-delay: 1 # ── AI Crawlers (used by AI companies for training / search) ────────────────── # Block only if you don't want AI training; these lines ALLOW AI crawling # Remove or add Disallow to restrict specific AI crawlers User-agent: GPTBot Allow: / User-agent: ChatGPT-User Allow: / User-agent: anthropic-ai Allow: / User-agent: Claude-Web Allow: / User-agent: CCBot Allow: / User-agent: OAI-SearchBot Allow: / User-agent: PerplexityBot Allow: / User-agent: Applebot Allow: / User-agent: Applebot-Extended Allow: / User-agent: Bytespider Allow: / Crawl-delay: 1 User-agent: PetalBot Allow: / User-agent: cohere-ai Allow: / User-agent: Diffbot Allow: / User-agent: FacebookBot Allow: / # ── Block scrapers that ignore crawl-delay ──────────────────────────────────── User-agent: AhrefsBot Disallow: / User-agent: SemrushBot Disallow: / User-agent: DotBot Disallow: / User-agent: MJ12bot Disallow: / # ── Host directive (mirror guidance for some crawlers) ──────────────────────── # Host: cnglobalsourcing.com