Search

Matched domain: arbisoft.com

IP = 188.245.123.186

robots.txt

# robots.txt for arbisoft.com
# Last updated: 2026-06-05
#
# llms.txt:      https://arbisoft.com/llms.txt
# llms-full.txt: https://arbisoft.com/llms-full.txt

# ============================================================
# GENERAL CRAWLERS
# ============================================================

User-Agent: *
Allow: /
Disallow: /lp/*

# URL parameters: NOT blocked here intentionally.
# Canonical tags on each page are the correct mechanism for
# telling Google which URL is authoritative (clean URL without
# UTM/tracking params). robots.txt Disallow for parameters
# blocks crawling but not indexing — which can create
# "Discovered – currently not indexed" issues in Search Console.
# Ensure  is set on every page.

# ============================================================
# GOOGLE ADS BOT (must keep for ad landing pages)
# ============================================================

User-Agent: AdsBot-Google
Allow: /lp/*

# ============================================================
# SPAM & BANDWIDTH-HEAVY CRAWLERS — BLOCKED
# These bots contribute nothing to search rankings or LLM
# training and waste significant crawl budget / server load
# ============================================================

# Majestic SEO (very aggressive, high bandwidth)
User-Agent: MJ12bot
Disallow: /

# Moz
User-Agent: DotBot
Disallow: /

# BLEXBot (aggressive link scraper)
User-Agent: BLEXBot
Disallow: /

# DataForSEO (scraper-as-a-service)
User-Agent: DataForSeoBot
Disallow: /

# SplitSignal (SEO crawler)
User-Agent: SplitSignalBot
Disallow: /

# MegaIndex (Russian SEO crawler)
User-Agent: MegaIndex
Disallow: /

# Proximic (contextual ad scraper)
User-Agent: Proximic
Disallow: /

# PetalBot (Huawei — aggressive, not relevant market)
User-Agent: PetalBot
Disallow: /

# Bytespider (ByteDance/TikTok — opaque training crawler)
User-Agent: Bytespider
Disallow: /

# SeekportBot
User-Agent: SeekportBot
Disallow: /

# Baiduspider (Chinese market — not Arbisoft target)
User-Agent: Baiduspider
Disallow: /

# YandexBot (Russian market — not Arbisoft target)
User-Agent: YandexBot
Disallow: /

# 008 crawler (known spam bot)
User-Agent: 008
Disallow: /

# AhrefsBot (very high crawl volume, bandwidth cost)
User-Agent: AhrefsBot
Disallow: /

# SemrushBot (very high crawl volume, bandwidth cost)
User-Agent: SemrushBot
Disallow: /

# ============================================================
# LLM & AI CRAWLERS — all explicitly allowed
# Full site access including llms.txt for best AI visibility
# ============================================================

# OpenAI / ChatGPT
User-Agent: GPTBot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# OpenAI browsing plugin
User-Agent: ChatGPT-User
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Anthropic / Claude
User-Agent: ClaudeBot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

User-Agent: Claude-Web
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

User-Agent: anthropic-ai
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Google Gemini / Bard / AI Overviews
User-Agent: Google-Extended
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Perplexity AI
User-Agent: PerplexityBot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Meta AI
User-Agent: Meta-ExternalAgent
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

User-Agent: Meta-ExternalFetcher
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Microsoft Copilot / Bing AI
User-Agent: Bingbot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

User-Agent: msnbot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# You.com
User-Agent: YouBot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Cohere AI
User-Agent: cohere-ai
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Common Crawl (used by many AI training datasets)
User-Agent: CCBot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Mistral AI
User-Agent: MistralBot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Grok / xAI
User-Agent: Grok
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Diffbot (used for knowledge graph training)
User-Agent: Diffbot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Applebot (Apple Intelligence / Siri)
User-Agent: Applebot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

User-Agent: Applebot-Extended
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# Amazon Alexa
User-Agent: Amazonbot
Allow: /
Allow: /llms.txt
Allow: /llms-full.txt

# ============================================================
# SITEMAPS
# ============================================================

Sitemap: https://arbisoft.com/sitemap.xml
# HTML sitemap (human-readable, crawled via normal link discovery):
# https://arbisoft.com/site-map

Look up this url in the url tool https://arbisoft.com/.well-known/acme-challenge: 404 text/html; charset=utf-8
https://arbisoft.com/.well-known/csvm: 404 text/html; charset=utf-8
https://arbisoft.com/.well-known/nostr.json: 404 text/html; charset=utf-8
https://arbisoft.com/.well-known/security.txt: 404 text/html; charset=utf-8
https://arbisoft.com/.well-known/traffic-advice: 404 text/html; charset=utf-8