<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Spicrawl blog</title>
    <link>https://spicrawl.com/blog</link>
    <atom:link href="https://spicrawl.com/blog/rss.xml" rel="self" type="application/rss+xml" />
    <description>Guides, tutorials and comparisons for scraping the web with AI agents.</description>
    <language>en</language>
    <item>
      <title>Cloudflare AI crawler blocking in 2026: what changed and what it means for your agent</title>
      <link>https://spicrawl.com/blog/cloudflare-blocking-ai-crawlers</link>
      <guid isPermaLink="true">https://spicrawl.com/blog/cloudflare-blocking-ai-crawlers</guid>
      <pubDate>Mon, 05 Oct 2026 00:00:00 GMT</pubDate>
      <category>Guides</category>
      <description>Cloudflare lets sites block Search, Agent and Training bots separately, with new defaults from 15 September 2026. What changed and how to scrape responsibly.</description>
    </item>
    <item>
      <title>Web scraping 403 Forbidden: causes and fixes</title>
      <link>https://spicrawl.com/blog/web-scraping-403-forbidden</link>
      <guid isPermaLink="true">https://spicrawl.com/blog/web-scraping-403-forbidden</guid>
      <pubDate>Mon, 05 Oct 2026 00:00:00 GMT</pubDate>
      <category>Tutorials</category>
      <description>Why a scraper gets a 403 Forbidden, how to tell the causes apart, and Python fixes that are fair to the site: headers, pacing, robots.txt and sessions.</description>
    </item>
    <item>
      <title>What is llms.txt? Examples and how to write one</title>
      <link>https://spicrawl.com/blog/llms-txt-explained</link>
      <guid isPermaLink="true">https://spicrawl.com/blog/llms-txt-explained</guid>
      <pubDate>Mon, 05 Oct 2026 00:00:00 GMT</pubDate>
      <category>Guides</category>
      <description>llms.txt is a plain file that points AI tools to your best pages. Learn the format, see a real example, and read what Google says about it.</description>
    </item>
    <item>
      <title>The best web scraping tools for AI agents in 2026</title>
      <link>https://spicrawl.com/blog/best-web-scraping-tools-for-ai-agents</link>
      <guid isPermaLink="true">https://spicrawl.com/blog/best-web-scraping-tools-for-ai-agents</guid>
      <pubDate>Thu, 01 Oct 2026 00:00:00 GMT</pubDate>
      <category>Comparisons</category>
      <description>An honest comparison of 10 web scraping tools for AI agents, from scraping APIs and MCP servers to open-source crawlers and cloud browsers.</description>
    </item>
    <item>
      <title>How to scrape a website with Claude Code</title>
      <link>https://spicrawl.com/blog/scrape-websites-with-claude-code</link>
      <guid isPermaLink="true">https://spicrawl.com/blog/scrape-websites-with-claude-code</guid>
      <pubDate>Thu, 01 Oct 2026 00:00:00 GMT</pubDate>
      <category>Tutorials</category>
      <description>Connect Claude Code to Spicrawl's hosted MCP server, read live pages as Markdown, and handle JavaScript, logins, costs and errors.</description>
    </item>
    <item>
      <title>HTML vs Markdown for LLMs: which uses fewer tokens?</title>
      <link>https://spicrawl.com/blog/html-vs-markdown-for-llms</link>
      <guid isPermaLink="true">https://spicrawl.com/blog/html-vs-markdown-for-llms</guid>
      <pubDate>Thu, 01 Oct 2026 00:00:00 GMT</pubDate>
      <category>Guides</category>
      <description>We measured six public pages with tiktoken: HTML and Spicrawl Markdown token counts, the method, what conversion keeps and loses, and when HTML wins.</description>
    </item>
  </channel>
</rss>
