{
  "created": "1755500000000",
  "updated": "1755500000000",
  "name": "Scrape.do — Bulk URL Scraper",
  "description": "Scrapes a list of URLs through the Scrape.do API on a schedule and extracts page title + status for each one. Anti-bot bypass, proxy rotation, CAPTCHA solving and JS rendering are handled by Scrape.do.",
  "tags": ["scraping", "scrape.do", "http"],
  "pieces": [
    "@activepieces/piece-schedule",
    "@activepieces/piece-http"
  ],
  "template": {
    "displayName": "Scrape.do — Bulk URL Scraper",
    "trigger": {
      "name": "trigger",
      "valid": true,
      "displayName": "Every Day",
      "type": "PIECE_TRIGGER",
      "settings": {
        "pieceName": "@activepieces/piece-schedule",
        "pieceVersion": "~0.1.5",
        "pieceType": "OFFICIAL",
        "packageType": "REGISTRY",
        "triggerName": "every_day",
        "input": {
          "hour_of_the_day": 9,
          "timezone": "UTC"
        },
        "inputUiInfo": {
          "customizedInputs": {}
        }
      },
      "nextAction": {
        "name": "step_1",
        "type": "CODE",
        "valid": true,
        "displayName": "Build URL List",
        "settings": {
          "input": {},
          "sourceCode": {
            "code": "export const code = async (inputs) => {\n  // Edit this list — or replace this step with a Google Sheets 'Get Rows' action\n  // and map the URL column here instead.\n  const urls = [\n    'https://books.toscrape.com/catalogue/page-1.html',\n    'https://books.toscrape.com/catalogue/page-2.html',\n    'https://books.toscrape.com/catalogue/page-3.html',\n  ];\n\n  return urls.map((url) => ({ url }));\n};",
            "packageJson": "{\n  \"dependencies\": {}\n}"
          },
          "inputUiInfo": {
            "customizedInputs": {}
          },
          "errorHandlingOptions": {
            "retryOnFailure": {
              "value": false
            },
            "continueOnFailure": {
              "value": false
            }
          }
        },
        "nextAction": {
          "name": "step_2",
          "type": "LOOP_ON_ITEMS",
          "valid": true,
          "displayName": "Loop Over URLs",
          "settings": {
            "items": "{{step_1}}",
            "inputUiInfo": {
              "customizedInputs": {}
            }
          },
          "firstLoopAction": {
            "name": "step_3",
            "type": "PIECE",
            "valid": true,
            "displayName": "Scrape.do Request",
            "settings": {
              "pieceName": "@activepieces/piece-http",
              "pieceVersion": "~0.5.0",
              "pieceType": "OFFICIAL",
              "packageType": "REGISTRY",
              "actionName": "send_request",
              "input": {
                "method": "GET",
                "url": "https://api.scrape.do/",
                "queryParams": {
                  "token": "YOUR_SCRAPEDO_TOKEN",
                  "url": "{{step_2.item.url}}"
                },
                "headers": {},
                "failsafe": true,
                "timeout": 120
              },
              "inputUiInfo": {
                "customizedInputs": {}
              },
              "errorHandlingOptions": {
                "retryOnFailure": {
                  "value": true
                },
                "continueOnFailure": {
                  "value": true
                }
              }
            },
            "nextAction": {
              "name": "step_4",
              "type": "CODE",
              "valid": true,
              "displayName": "Extract Fields",
              "settings": {
                "input": {
                  "url": "{{step_2.item.url}}",
                  "status": "{{step_3.status}}",
                  "html": "{{step_3.body}}"
                },
                "sourceCode": {
                  "code": "export const code = async (inputs) => {\n  const html = String(inputs.html || '');\n\n  const title = (html.match(/<title[^>]*>([\\s\\S]*?)<\\/title>/i) || [])[1] || '';\n  const h1 = (html.match(/<h1[^>]*>([\\s\\S]*?)<\\/h1>/i) || [])[1] || '';\n\n  const clean = (s) => s.replace(/<[^>]*>/g, '').replace(/\\s+/g, ' ').trim();\n\n  return {\n    url: inputs.url,\n    status: inputs.status,\n    title: clean(title),\n    h1: clean(h1),\n    html_length: html.length,\n    scraped_at: new Date().toISOString(),\n  };\n};",
                  "packageJson": "{\n  \"dependencies\": {}\n}"
                },
                "inputUiInfo": {
                  "customizedInputs": {}
                },
                "errorHandlingOptions": {
                  "retryOnFailure": {
                    "value": false
                  },
                  "continueOnFailure": {
                    "value": true
                  }
                }
              }
            }
          }
        }
      }
    },
    "valid": true
  },
  "blogUrl": ""
}
