User-agent: * Allow: / # Public search indexing and attributed citations are welcome, subject to these # exclusions and server access controls. See /data-policy.php and /llms.txt. # This documentation does not change existing training-crawler permissions. Disallow: /data/ Disallow: /tests/ Disallow: /history.php Disallow: /quality.php Disallow: /urls.php Disallow: /datadump.php Disallow: /login_etfllama_adminpage.php Disallow: /crawl.php Disallow: /crawl_one.php Disallow: /crawl_usa.php Disallow: /crawler_main.php Disallow: /migrate_crawl_urls.php Disallow: /migrate_to_pg.php Disallow: /migrate_etf_id.php Disallow: /urls_export.php Disallow: /config.php Disallow: /db_functions.php Disallow: /auth_check.php Disallow: /api/ # Google-Extended is a robots.txt product token used for Gemini training and # grounding. Googlebot itself continues to use the general rules above. User-agent: Google-Extended Allow: / # Dataset/training crawlers that are not part of the approved Google, # OpenAI, Anthropic, or search-engine groups. User-agent: Amazonbot Disallow: / User-agent: Applebot-Extended Disallow: / User-agent: Bytespider Disallow: / User-agent: CCBot Disallow: / User-agent: cohere-ai Disallow: / User-agent: Diffbot Disallow: / User-agent: FacebookBot Disallow: / User-agent: ImagesiftBot Disallow: / User-agent: meta-externalagent Disallow: / User-agent: omgili Disallow: / User-agent: omgilibot Disallow: / User-agent: PanguBot Disallow: / User-agent: Timpibot Disallow: / Sitemap: https://etfllama.com/sitemap.xml