{
 "operator": "LAION / img2dataset",
 "slug": "laion",
 "docs": "https://github.com/rom1504/img2dataset",
 "crawler_count": 2,
 "crawlers": [
  {
   "slug": "img2dataset",
   "name": "img2dataset",
   "operator": "LAION / img2dataset",
   "operator_slug": "laion",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "img2dataset",
   "user_agent_substring": "img2dataset",
   "user_agent_example": "img2dataset",
   "respects_robots_txt": "documented",
   "respects_robots_txt_label": "obeys robots.txt (documented)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "The tool used to turn image-URL lists such as LAION's into downloaded training sets. It is run by whoever is building a dataset, not by a single operator.",
   "cost_of_blocking": "Your images are skipped when someone materialises an image-text dataset that references them.",
   "operator_docs": "https://github.com/rom1504/img2dataset",
   "html_url": "https://www.pathwren.workers.dev/crawler/img2dataset.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/img2dataset.json",
   "last_reviewed": "2026-09-03"
  },
  {
   "slug": "laiondownloader",
   "name": "LAIONDownloader",
   "operator": "LAION / img2dataset",
   "operator_slug": "laion",
   "category": "dataset",
   "category_label": "Corpus and dataset builders",
   "robots_token": "LAIONDownloader",
   "user_agent_substring": "LAIONDownloader",
   "user_agent_example": "LAIONDownloader",
   "respects_robots_txt": "by-design-no",
   "respects_robots_txt_label": "not governed by robots.txt (user-initiated, by operator policy)",
   "verification_method": "none",
   "verification_label": "no published verification method",
   "published_ip_ranges_url": null,
   "ip_ranges_endpoint": null,
   "ipv4_prefix_count": 0,
   "ipv6_prefix_count": 0,
   "what_it_is": "LAION's downloader, used to materialise the image and text datasets the non-profit publishes for machine-learning research. LAION's own FAQ is the source for its robots.txt position.",
   "cost_of_blocking": "Your media is skipped when an open research dataset is built from URL lists. Once a dataset is published, a later block does not remove you from it.",
   "operator_docs": "https://laion.ai/faq/",
   "html_url": "https://www.pathwren.workers.dev/crawler/laiondownloader.html",
   "json_url": "https://www.pathwren.workers.dev/crawler/laiondownloader.json",
   "last_reviewed": "2026-09-03"
  }
 ],
 "ip_range_endpoints": []
}