## Description Migrates Firecrawl from the v1 to the v2 API. `Captain::Tools::FirecrawlService` now targets `api.firecrawl.dev/v2`, with the request body updated to match the v2 schema. > Disclosure: I work at Firecrawl. Fixes # (n/a) ## Type of change - [x] Bug fix (non-breaking change which fixes an issue) ## How Has This Been Tested? Updated `spec/enterprise/services/captain/tools/firecrawl_service_spec.rb` to assert the v2 endpoint and request body. ## Checklist: - [x] My code follows the style guidelines of this project - [x] I have performed a self-review of my code - [x] My changes generate no new warnings - [x] I have added tests that prove my fix is effective or that my feature works - [x] New and existing unit tests pass locally with my changes --------- Co-authored-by: Aakash Bakhle <48802744+aakashb95@users.noreply.github.com>
66 lines
1.5 KiB
Ruby
66 lines
1.5 KiB
Ruby
class Captain::Tools::FirecrawlService
|
|
BASE_URL = 'https://api.firecrawl.dev/v2'.freeze
|
|
FIRECRAWL_EXCLUDE_TAGS = %w[iframe .sidebar .cookie-banner [role=navigation] [role=banner] [role=contentinfo]].freeze
|
|
|
|
def self.configured?
|
|
InstallationConfig.find_by(name: 'CAPTAIN_FIRECRAWL_API_KEY')&.value
|
|
.present?
|
|
end
|
|
|
|
def initialize
|
|
@api_key = InstallationConfig.find_by!(name: 'CAPTAIN_FIRECRAWL_API_KEY').value
|
|
raise 'Missing API key' if @api_key.blank?
|
|
end
|
|
|
|
def perform(url, webhook_url, crawl_limit = 10)
|
|
HTTParty.post(
|
|
"#{BASE_URL}/crawl",
|
|
body: crawl_payload(url, webhook_url, crawl_limit),
|
|
headers: headers
|
|
)
|
|
rescue StandardError => e
|
|
raise "Failed to crawl URL: #{e.message}"
|
|
end
|
|
|
|
def scrape(url)
|
|
HTTParty.post(
|
|
"#{BASE_URL}/scrape",
|
|
body: scrape_payload(url),
|
|
headers: headers
|
|
)
|
|
end
|
|
|
|
private
|
|
|
|
def crawl_payload(url, webhook_url, crawl_limit)
|
|
{
|
|
url: url,
|
|
maxDiscoveryDepth: 50,
|
|
sitemap: 'include',
|
|
limit: crawl_limit,
|
|
webhook: { url: webhook_url },
|
|
scrapeOptions: scrape_options
|
|
}.to_json
|
|
end
|
|
|
|
def scrape_payload(url)
|
|
{ url: url }.merge(scrape_options).to_json
|
|
end
|
|
|
|
def scrape_options
|
|
{
|
|
onlyMainContent: true,
|
|
formats: ['markdown'],
|
|
excludeTags: FIRECRAWL_EXCLUDE_TAGS,
|
|
maxAge: 0
|
|
}
|
|
end
|
|
|
|
def headers
|
|
{
|
|
'Authorization' => "Bearer #{@api_key}",
|
|
'Content-Type' => 'application/json'
|
|
}
|
|
end
|
|
end
|