From 871f2f4d56516d4fc07b6a242dab60872673995b Mon Sep 17 00:00:00 2001 From: Shivam Mishra Date: Wed, 8 Apr 2026 10:47:54 +0530 Subject: [PATCH 01/17] fix: harden fetching on upload endpoint (#14012) --- Gemfile | 2 + Gemfile.lock | 2 + .../api/v1/accounts/upload_controller.rb | 42 +-- config/locales/en.yml | 8 + lib/safe_fetch.rb | 98 +++++++ .../api/v1/upload_controller_spec.rb | 113 +++++++- spec/lib/safe_fetch_spec.rb | 258 ++++++++++++++++++ 7 files changed, 494 insertions(+), 29 deletions(-) create mode 100644 lib/safe_fetch.rb create mode 100644 spec/lib/safe_fetch_spec.rb diff --git a/Gemfile b/Gemfile index 01c7a9f83..a5068e765 100644 --- a/Gemfile +++ b/Gemfile @@ -40,6 +40,8 @@ gem 'json_refs' gem 'rack-attack', '>= 6.7.0' # a utility tool for streaming, flexible and safe downloading of remote files gem 'down' +# SSRF-safe URL fetching +gem 'ssrf_filter', '~> 1.5' # authentication type to fetch and send mail over oauth2.0 gem 'gmail_xoauth' # Lock net-smtp to 0.3.4 to avoid issues with gmail_xoauth2 diff --git a/Gemfile.lock b/Gemfile.lock index 74ea4d82d..b77e5880f 100644 --- a/Gemfile.lock +++ b/Gemfile.lock @@ -942,6 +942,7 @@ GEM activesupport (>= 5.2) sprockets (>= 3.0.0) squasher (0.7.2) + ssrf_filter (1.5.0) stackprof (0.2.25) statsd-ruby (1.5.0) stripe (18.0.1) @@ -1158,6 +1159,7 @@ DEPENDENCIES spring spring-watcher-listen squasher + ssrf_filter (~> 1.5) stackprof stripe (~> 18.0) telephone_number diff --git a/app/controllers/api/v1/accounts/upload_controller.rb b/app/controllers/api/v1/accounts/upload_controller.rb index 479d8ae1b..bf20bc6ff 100644 --- a/app/controllers/api/v1/accounts/upload_controller.rb +++ b/app/controllers/api/v1/accounts/upload_controller.rb @@ -5,7 +5,7 @@ class Api::V1::Accounts::UploadController < Api::V1::Accounts::BaseController elsif params[:external_url].present? create_from_url else - render_error('No file or URL provided', :unprocessable_entity) + render_error(I18n.t('errors.upload.missing_input'), :unprocessable_entity) end render_success(result) if result.is_a?(ActiveStorage::Blob) @@ -19,35 +19,21 @@ class Api::V1::Accounts::UploadController < Api::V1::Accounts::BaseController end def create_from_url - uri = parse_uri(params[:external_url]) - return if performed? - - fetch_and_process_file_from_uri(uri) - end - - def parse_uri(url) - uri = URI.parse(url) - validate_uri(uri) - uri - rescue URI::InvalidURIError, SocketError - render_error('Invalid URL provided', :unprocessable_entity) - nil - end - - def validate_uri(uri) - raise URI::InvalidURIError unless uri.is_a?(URI::HTTP) || uri.is_a?(URI::HTTPS) - end - - def fetch_and_process_file_from_uri(uri) - uri.open do |file| - create_and_save_blob(file, File.basename(uri.path), file.content_type) + SafeFetch.fetch(params[:external_url].to_s) do |result| + create_and_save_blob(result.tempfile, result.filename, result.content_type) end - rescue OpenURI::HTTPError => e - render_error("Failed to fetch file from URL: #{e.message}", :unprocessable_entity) - rescue SocketError - render_error('Invalid URL provided', :unprocessable_entity) + rescue SafeFetch::HttpError => e + render_error(I18n.t('errors.upload.fetch_failed_with_message', message: e.message), :unprocessable_entity) + rescue SafeFetch::FetchError + render_error(I18n.t('errors.upload.fetch_failed'), :unprocessable_entity) + rescue SafeFetch::FileTooLargeError + render_error(I18n.t('errors.upload.file_too_large'), :unprocessable_entity) + rescue SafeFetch::UnsupportedContentTypeError + render_error(I18n.t('errors.upload.unsupported_content_type'), :unprocessable_entity) + rescue SafeFetch::Error + render_error(I18n.t('errors.upload.invalid_url'), :unprocessable_entity) rescue StandardError - render_error('An unexpected error occurred', :internal_server_error) + render_error(I18n.t('errors.upload.unexpected'), :internal_server_error) end def create_and_save_blob(io, filename, content_type) diff --git a/config/locales/en.yml b/config/locales/en.yml index e6308c43c..1cb3c4d12 100644 --- a/config/locales/en.yml +++ b/config/locales/en.yml @@ -66,6 +66,14 @@ en: not_found: Assignment policy not found attachments: invalid: Invalid attachment + upload: + missing_input: 'No file or URL provided' + invalid_url: 'Invalid URL provided' + fetch_failed: 'Failed to fetch file from URL' + fetch_failed_with_message: 'Failed to fetch file from URL: %{message}' + file_too_large: 'File exceeds the maximum allowed size' + unsupported_content_type: 'File type not supported (only images and videos are allowed)' + unexpected: 'An unexpected error occurred' saml: feature_not_enabled: SAML feature not enabled for this account sso_not_enabled: SAML SSO is not enabled for this installation diff --git a/lib/safe_fetch.rb b/lib/safe_fetch.rb new file mode 100644 index 000000000..e6635c9c3 --- /dev/null +++ b/lib/safe_fetch.rb @@ -0,0 +1,98 @@ +require 'ssrf_filter' + +module SafeFetch + DEFAULT_ALLOWED_CONTENT_TYPE_PREFIXES = %w[image/ video/].freeze + DEFAULT_OPEN_TIMEOUT = 2 + DEFAULT_READ_TIMEOUT = 20 + DEFAULT_MAX_BYTES_FALLBACK_MB = 40 + + Result = Data.define(:tempfile, :filename, :content_type) + + class Error < StandardError; end + class InvalidUrlError < Error; end + class UnsafeUrlError < Error; end + class FetchError < Error; end + class HttpError < Error; end + class FileTooLargeError < Error; end + class UnsupportedContentTypeError < Error; end + + def self.fetch(url, + max_bytes: nil, + allowed_content_type_prefixes: DEFAULT_ALLOWED_CONTENT_TYPE_PREFIXES) + raise ArgumentError, 'block required' unless block_given? + + effective_max_bytes = max_bytes || default_max_bytes + uri = parse_and_validate_url!(url) + filename = filename_for(uri) + tempfile = Tempfile.new('chatwoot-safe-fetch', binmode: true) + + response = stream_to_tempfile(url, tempfile, effective_max_bytes, allowed_content_type_prefixes) + raise HttpError, "#{response.code} #{response.message}" unless response.is_a?(Net::HTTPSuccess) + + tempfile.rewind + yield Result.new(tempfile: tempfile, filename: filename, content_type: response['content-type']) + rescue SsrfFilter::InvalidUriScheme, URI::InvalidURIError => e + raise InvalidUrlError, e.message + rescue SsrfFilter::Error, Resolv::ResolvError => e + raise UnsafeUrlError, e.message + rescue Net::OpenTimeout, Net::ReadTimeout, SocketError, OpenSSL::SSL::SSLError => e + raise FetchError, e.message + ensure + tempfile&.close! + end + + class << self + private + + def stream_to_tempfile(url, tempfile, max_bytes, allowed_content_type_prefixes) + response = nil + bytes_written = 0 + + SsrfFilter.get( + url, + http_options: { open_timeout: DEFAULT_OPEN_TIMEOUT, read_timeout: DEFAULT_READ_TIMEOUT } + ) do |res| + response = res + next unless res.is_a?(Net::HTTPSuccess) + + unless allowed_content_type?(res['content-type'], allowed_content_type_prefixes) + raise UnsupportedContentTypeError, "content-type not allowed: #{res['content-type']}" + end + + res.read_body do |chunk| + bytes_written += chunk.bytesize + raise FileTooLargeError, "exceeded #{max_bytes} bytes" if bytes_written > max_bytes + + tempfile.write(chunk) + end + end + + response + end + + def filename_for(uri) + File.basename(uri.path).presence || "download-#{Time.current.to_i}-#{SecureRandom.hex(4)}" + end + + def default_max_bytes + limit_mb = GlobalConfigService.load('MAXIMUM_FILE_UPLOAD_SIZE', DEFAULT_MAX_BYTES_FALLBACK_MB).to_i + limit_mb = DEFAULT_MAX_BYTES_FALLBACK_MB if limit_mb <= 0 + limit_mb.megabytes + end + + def parse_and_validate_url!(url) + uri = URI.parse(url) + raise InvalidUrlError, 'scheme must be http or https' unless uri.is_a?(URI::HTTP) || uri.is_a?(URI::HTTPS) + raise InvalidUrlError, 'missing host' if uri.host.blank? + + uri + end + + def allowed_content_type?(value, prefixes) + mime = value.to_s.split(';').first&.strip&.downcase + return false if mime.blank? + + prefixes.any? { |prefix| mime.start_with?(prefix) } + end + end +end diff --git a/spec/controllers/api/v1/upload_controller_spec.rb b/spec/controllers/api/v1/upload_controller_spec.rb index 93ef28dd8..2878c2a5e 100644 --- a/spec/controllers/api/v1/upload_controller_spec.rb +++ b/spec/controllers/api/v1/upload_controller_spec.rb @@ -39,6 +39,11 @@ RSpec.describe 'Api::V1::Accounts::UploadController', type: :request do let(:valid_external_url) { 'http://example.com/image.jpg' } before do + allow(Resolv).to receive(:getaddresses).and_call_original + allow(Resolv).to receive(:getaddresses).with('example.com').and_return(['93.184.216.34']) + allow(Resolv).to receive(:getaddresses).with('error.example.com').and_return(['93.184.216.34']) + allow(Resolv).to receive(:getaddresses).with('nonexistent.example.com').and_return(['93.184.216.34']) + stub_request(:get, valid_external_url) .to_return(status: 200, body: File.new(Rails.root.join('spec/assets/avatar.png')), headers: { 'Content-Type' => 'image/png' }) end @@ -82,7 +87,7 @@ RSpec.describe 'Api::V1::Accounts::UploadController', type: :request do params: { external_url: 'http://nonexistent.example.com' } expect(response).to have_http_status(:unprocessable_entity) - expect(response.parsed_body['error']).to eq('Invalid URL provided') + expect(response.parsed_body['error']).to eq('Failed to fetch file from URL') end it 'handles HTTP errors' do @@ -96,6 +101,112 @@ RSpec.describe 'Api::V1::Accounts::UploadController', type: :request do expect(response).to have_http_status(:unprocessable_entity) expect(response.parsed_body['error']).to start_with('Failed to fetch file from URL') end + + it 'rejects oversized responses with a file-size message' do + stub_request(:get, valid_external_url) + .to_return(status: 200, + body: 'x' * (41 * 1024 * 1024), + headers: { 'Content-Type' => 'image/png' }) + + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: valid_external_url } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('File exceeds the maximum allowed size') + end + + it 'rejects unsupported content types with a file-type message' do + stub_request(:get, valid_external_url) + .to_return(status: 200, + body: '', + headers: { 'Content-Type' => 'text/html' }) + + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: valid_external_url } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('File type not supported (only images and videos are allowed)') + end + + context 'with SSRF attack vectors' do + it 'blocks requests to private IP ranges (10.x.x.x)' do + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: 'http://10.0.0.1/secret' } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('Invalid URL provided') + end + + it 'blocks requests to private IP ranges (172.16.x.x)' do + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: 'http://172.16.0.1/secret' } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('Invalid URL provided') + end + + it 'blocks requests to private IP ranges (192.168.x.x)' do + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: 'http://192.168.1.1/secret' } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('Invalid URL provided') + end + + it 'blocks requests to loopback addresses' do + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: 'http://127.0.0.1/secret' } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('Invalid URL provided') + end + + it 'blocks requests to AWS metadata service (169.254.169.254)' do + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: 'http://169.254.169.254/latest/meta-data/iam/security-credentials/' } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('Invalid URL provided') + end + + it 'blocks requests to localhost' do + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: 'http://localhost/secret' } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('Invalid URL provided') + end + + it 'blocks requests to .local domains' do + allow(Resolv).to receive(:getaddresses).with('server.local').and_return(['192.168.1.100']) + + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: 'http://server.local/secret' } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('Invalid URL provided') + end + + it 'blocks DNS rebinding attacks (hostname resolving to private IP)' do + allow(Resolv).to receive(:getaddresses).with('evil.attacker.com').and_return(['10.0.0.1']) + + post upload_url, + headers: user.create_new_auth_token, + params: { external_url: 'http://evil.attacker.com/secret' } + + expect(response).to have_http_status(:unprocessable_entity) + expect(response.parsed_body['error']).to eq('Invalid URL provided') + end + end end it 'returns an error when no file or URL is provided' do diff --git a/spec/lib/safe_fetch_spec.rb b/spec/lib/safe_fetch_spec.rb new file mode 100644 index 000000000..70f9b05de --- /dev/null +++ b/spec/lib/safe_fetch_spec.rb @@ -0,0 +1,258 @@ +require 'rails_helper' + +# `SafeFetch.fetch` is a custom method that requires a block (it yields a Result); +# it is NOT `Hash#fetch`, so RuboCop's autocorrect to `fetch(url, nil)` would break the API. +# rubocop:disable Style/RedundantFetchBlock +RSpec.describe SafeFetch do + let(:url) { 'http://example.com/image.png' } + + before do + allow(Resolv).to receive(:getaddresses).and_call_original + allow(Resolv).to receive(:getaddresses).with('example.com').and_return(['93.184.216.34']) + end + + describe '.fetch' do + context 'with a valid public URL serving an image' do + before do + stub_request(:get, url).to_return( + status: 200, + body: File.new(Rails.root.join('spec/assets/avatar.png')), + headers: { 'Content-Type' => 'image/png' } + ) + end + + it 'yields a Result with tempfile, filename, and content_type' do + described_class.fetch(url) do |result| + expect(result.tempfile).to be_a(Tempfile) + expect(result.filename).to eq('image.png') + expect(result.content_type).to eq('image/png') + expect(result.tempfile.size).to be > 0 + end + end + + it 'closes the tempfile after the block returns' do + captured = nil + described_class.fetch(url) { |result| captured = result.tempfile } + expect(captured.closed?).to be true + end + + it 'closes the tempfile even when the block raises' do + captured = nil + expect do + described_class.fetch(url) do |result| + captured = result.tempfile + raise 'boom' + end + end.to raise_error('boom') + expect(captured.closed?).to be true + end + + it 'defaults the filename to a unique "download--" when the URL has no path' do + bare_url = 'http://example.com' + stub_request(:get, bare_url).to_return( + status: 200, + body: File.new(Rails.root.join('spec/assets/avatar.png')), + headers: { 'Content-Type' => 'image/png' } + ) + + described_class.fetch(bare_url) do |result| + expect(result.filename).to match(/\Adownload-\d+-[a-f0-9]{8}\z/) + end + end + + it 'requires a block' do + expect { described_class.fetch(url) }.to raise_error(ArgumentError, /block required/) + end + end + + context 'with URL validation' do + it 'raises InvalidUrlError for javascript: URLs' do + expect { described_class.fetch('javascript:alert(1)') { nil } } + .to raise_error(SafeFetch::InvalidUrlError) + end + + it 'raises InvalidUrlError for mailto: URLs' do + expect { described_class.fetch('mailto:test@example.com') { nil } } + .to raise_error(SafeFetch::InvalidUrlError) + end + + it 'raises InvalidUrlError for data: URLs' do + expect { described_class.fetch('data:text/html,') { nil } } + .to raise_error(SafeFetch::InvalidUrlError) + end + + it 'raises InvalidUrlError for ftp: URLs' do + expect { described_class.fetch('ftp://example.com/file') { nil } } + .to raise_error(SafeFetch::InvalidUrlError) + end + + it 'raises InvalidUrlError for malformed URLs' do + expect { described_class.fetch('not_a_url') { nil } } + .to raise_error(SafeFetch::InvalidUrlError) + end + + it 'raises InvalidUrlError when host is missing' do + expect { described_class.fetch('http:///path') { nil } } + .to raise_error(SafeFetch::InvalidUrlError, /missing host/) + end + end + + context 'with SSRF protection (integration with ssrf_filter)' do + it 'raises UnsafeUrlError for private IP literals (10.x.x.x)' do + expect { described_class.fetch('http://10.0.0.1/secret') { nil } } + .to raise_error(SafeFetch::UnsafeUrlError) + end + + it 'raises UnsafeUrlError for loopback addresses' do + expect { described_class.fetch('http://127.0.0.1/secret') { nil } } + .to raise_error(SafeFetch::UnsafeUrlError) + end + + it 'raises UnsafeUrlError for AWS metadata IP (169.254.169.254)' do + expect { described_class.fetch('http://169.254.169.254/latest/meta-data/') { nil } } + .to raise_error(SafeFetch::UnsafeUrlError) + end + + it 'raises UnsafeUrlError when hostname resolves to a private IP (DNS rebinding)' do + allow(Resolv).to receive(:getaddresses).with('evil.example.com').and_return(['10.0.0.1']) + expect { described_class.fetch('http://evil.example.com/secret') { nil } } + .to raise_error(SafeFetch::UnsafeUrlError) + end + end + + context 'with content-type allowlist' do + it 'rejects text/html responses' do + stub_request(:get, url).to_return( + status: 200, + body: '', + headers: { 'Content-Type' => 'text/html' } + ) + + expect { described_class.fetch(url) { nil } } + .to raise_error(SafeFetch::UnsupportedContentTypeError) + end + + it 'rejects application/octet-stream responses' do + stub_request(:get, url).to_return( + status: 200, + body: 'x', + headers: { 'Content-Type' => 'application/octet-stream' } + ) + + expect { described_class.fetch(url) { nil } } + .to raise_error(SafeFetch::UnsupportedContentTypeError) + end + + it 'allows video/mp4 responses' do + stub_request(:get, url).to_return( + status: 200, + body: File.new(Rails.root.join('spec/assets/avatar.png')), + headers: { 'Content-Type' => 'video/mp4' } + ) + + expect { described_class.fetch(url) { nil } }.not_to raise_error + end + + it 'strips charset/boundary parameters before comparing' do + stub_request(:get, url).to_return( + status: 200, + body: 'x', + headers: { 'Content-Type' => 'image/png; charset=binary' } + ) + + expect { described_class.fetch(url) { nil } }.not_to raise_error + end + + it 'rejects when the content-type header is missing' do + stub_request(:get, url).to_return(status: 200, body: 'x', headers: {}) + + expect { described_class.fetch(url) { nil } } + .to raise_error(SafeFetch::UnsupportedContentTypeError) + end + end + + context 'with body size cap' do + it 'honours a custom max_bytes argument' do + stub_request(:get, url).to_return( + status: 200, + body: 'xxxxx', + headers: { 'Content-Type' => 'image/png' } + ) + + expect { described_class.fetch(url, max_bytes: 2) { nil } } + .to raise_error(SafeFetch::FileTooLargeError) + end + + it 'reads the default cap from GlobalConfigService MAXIMUM_FILE_UPLOAD_SIZE (matching Attachment#validate_file_size)' do + allow(GlobalConfigService).to receive(:load).and_call_original + allow(GlobalConfigService).to receive(:load).with('MAXIMUM_FILE_UPLOAD_SIZE', 40).and_return('1') + + oversize = 'x' * (1.megabyte + 1) + stub_request(:get, url).to_return( + status: 200, + body: oversize, + headers: { 'Content-Type' => 'image/png' } + ) + + expect { described_class.fetch(url) { nil } } + .to raise_error(SafeFetch::FileTooLargeError) + end + + it 'falls back to 40 MB when GlobalConfigService returns a non-positive value' do + allow(GlobalConfigService).to receive(:load).and_call_original + allow(GlobalConfigService).to receive(:load).with('MAXIMUM_FILE_UPLOAD_SIZE', 40).and_return('-10') + + # 1 MB body should pass under the 40 MB fallback + stub_request(:get, url).to_return( + status: 200, + body: 'x' * 1.megabyte, + headers: { 'Content-Type' => 'image/png' } + ) + + expect { described_class.fetch(url) { nil } }.not_to raise_error + end + + it 'allows uploads between the old hardcoded 10 MB and the configured limit (regression check)' do + # Default config is 40 MB; a 15 MB upload must succeed. + # This is the exact regression scenario: with the old hardcoded 10 MB cap, + # this would have failed even though direct file uploads of the same size succeed. + allow(GlobalConfigService).to receive(:load).and_call_original + allow(GlobalConfigService).to receive(:load).with('MAXIMUM_FILE_UPLOAD_SIZE', 40).and_return('40') + + stub_request(:get, url).to_return( + status: 200, + body: 'x' * (15 * 1024 * 1024), + headers: { 'Content-Type' => 'image/png' } + ) + + expect { described_class.fetch(url) { nil } }.not_to raise_error + end + end + + context 'with network failures' do + it 'maps Net::ReadTimeout to FetchError' do + stub_request(:get, url).to_raise(Net::ReadTimeout) + + expect { described_class.fetch(url) { nil } } + .to raise_error(SafeFetch::FetchError) + end + + it 'maps SocketError to FetchError' do + stub_request(:get, url).to_raise(SocketError.new('connection refused')) + + expect { described_class.fetch(url) { nil } } + .to raise_error(SafeFetch::FetchError) + end + end + + context 'with non-2xx upstream responses' do + it 'raises HttpError with the status code in the message' do + stub_request(:get, url).to_return(status: 404, body: '', headers: {}) + + expect { described_class.fetch(url) { nil } } + .to raise_error(SafeFetch::HttpError, /404/) + end + end + end +end +# rubocop:enable Style/RedundantFetchBlock From e5107604a051d41839a1667431275221b0d236c6 Mon Sep 17 00:00:00 2001 From: Shivam Mishra Date: Wed, 8 Apr 2026 11:16:52 +0530 Subject: [PATCH 02/17] feat: account enrichment using context.dev [UPM-27] (#13978) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Account branding enrichment during signup This PR does the following ### Replace Firecrawl with Context.dev Switches the enterprise brand lookup from Firecrawl to Context.dev for better data quality, built-in caching, and automatic filtering of free/disposable email providers. The service interface changes from URL to email input to match Context.dev's email endpoint. OSS still falls back to basic HTML scraping with a normalized output shape across both paths. The enterprise path intentionally does not fall back to HTML scraping on failure — speed matters more than completeness. We want the user on the editable onboarding form fast, and a slow fallback scrape is worse than letting them fill it in. Requires `CONTEXT_DEV_API_KEY` in Super Admin → App Config. Without it, falls back to OSS HTML scraping. ### Add job to enrich account details After account creation, `Account::BrandingEnrichmentJob` looks up the signup email and pre-fills the account name, colors, logos, social links, and industry into `custom_attributes['brand_info']`. The job signals completion via a short-lived Redis key (30s TTL) + an ActionCable broadcast (`account.enrichment_completed`). The Redis key lets the frontend distinguish "still running" from "finished with no results." --- app/controllers/api/v1/accounts_controller.rb | 11 ++ app/jobs/account/branding_enrichment_job.rb | 32 +++ .../{concerns => }/social_link_parser.rb | 0 app/services/website_branding_service.rb | 84 ++++---- config/installation_config.yml | 7 + .../super_admin/app_configs_controller.rb | 6 +- .../enterprise/website_branding_service.rb | 123 ++++-------- lib/redis/redis_keys.rb | 3 + .../website_branding_service_spec.rb | 187 +++++++----------- .../services/website_branding_service_spec.rb | 85 ++++---- 10 files changed, 250 insertions(+), 288 deletions(-) create mode 100644 app/jobs/account/branding_enrichment_job.rb rename app/services/{concerns => }/social_link_parser.rb (100%) diff --git a/app/controllers/api/v1/accounts_controller.rb b/app/controllers/api/v1/accounts_controller.rb index 2d14fe7ca..7176d6e1b 100644 --- a/app/controllers/api/v1/accounts_controller.rb +++ b/app/controllers/api/v1/accounts_controller.rb @@ -30,6 +30,7 @@ class Api::V1::AccountsController < Api::BaseController locale: account_params[:locale], user: current_user ).perform + enqueue_branding_enrichment if @user # Authenticated users (dashboard "add account") and api_only signups # need the full response with account_id. API-only deployments have no @@ -69,6 +70,16 @@ class Api::V1::AccountsController < Api::BaseController private + def enqueue_branding_enrichment + return if account_params[:email].blank? + + Account::BrandingEnrichmentJob.perform_later(@account.id, account_params[:email]) + Redis::Alfred.set(format(Redis::Alfred::ACCOUNT_ONBOARDING_ENRICHMENT, account_id: @account.id), '1', ex: 30) + rescue StandardError => e + # Enrichment is optional — never let queue/Redis failures abort signup + ChatwootExceptionTracker.new(e).capture_exception + end + def ensure_account_name # ensure that account_name and user_full_name is present # this is becuase the account builder and the models validations are not triggered diff --git a/app/jobs/account/branding_enrichment_job.rb b/app/jobs/account/branding_enrichment_job.rb new file mode 100644 index 000000000..2898604ca --- /dev/null +++ b/app/jobs/account/branding_enrichment_job.rb @@ -0,0 +1,32 @@ +class Account::BrandingEnrichmentJob < ApplicationJob + queue_as :low + + def perform(account_id, email) + result = WebsiteBrandingService.new(email).perform + return if result.blank? + + account = Account.find(account_id) + account.name = result[:title] if result[:title].present? + account.custom_attributes['brand_info'] = result if account.custom_attributes['brand_info'].blank? + account.save! if account.changed? + ensure + finish_enrichment(account_id) + end + + private + + def finish_enrichment(account_id) + Redis::Alfred.delete(format(Redis::Alfred::ACCOUNT_ONBOARDING_ENRICHMENT, account_id: account_id)) + + account = Account.find(account_id) + if account.custom_attributes['onboarding_step'] == 'enrichment' + account.custom_attributes['onboarding_step'] = 'account_details' + account.save! + end + + user = account.administrators.first + return unless user + + ActionCableBroadcastJob.perform_later([user.pubsub_token], 'account.enrichment_completed', { account_id: account_id }) + end +end diff --git a/app/services/concerns/social_link_parser.rb b/app/services/social_link_parser.rb similarity index 100% rename from app/services/concerns/social_link_parser.rb rename to app/services/social_link_parser.rb diff --git a/app/services/website_branding_service.rb b/app/services/website_branding_service.rb index 89326e4b6..3592267ff 100644 --- a/app/services/website_branding_service.rb +++ b/app/services/website_branding_service.rb @@ -1,8 +1,15 @@ class WebsiteBrandingService include SocialLinkParser - def initialize(url) - @url = normalize_url(url) + attr_reader :http_status + + DATA_DEFAULTS = { description: nil, slogan: nil, phone: nil, address: nil, links: nil, stock: nil, industries: [], is_nsfw: false }.freeze + + def initialize(email) + @email = email + @domain = email.split('@').last&.downcase&.strip + @url = "https://#{@domain}" + @http_status = nil end def perform @@ -11,13 +18,14 @@ class WebsiteBrandingService links = extract_links(doc) - { - business_name: extract_business_name(doc), - language: extract_language(doc), - industry_category: nil, - social_handles: extract_social_from_links(links), - branding: extract_branding(doc) - } + DATA_DEFAULTS.merge({ + domain: @domain, + title: extract_title(doc), + colors: extract_colors(doc), + logos: extract_logos(doc), + socials: build_socials(links), + email: @email + }) rescue StandardError => e Rails.logger.error "[WebsiteBranding] #{e.message}" nil @@ -25,12 +33,9 @@ class WebsiteBrandingService private - def normalize_url(url) - url.match?(%r{\Ahttps?://}) ? url : "https://#{url}" - end - def fetch_page response = HTTParty.get(@url, follow_redirects: true, timeout: 15) + @http_status = response.code return nil unless response.success? Nokogiri::HTML(response.body) @@ -39,7 +44,7 @@ class WebsiteBrandingService nil end - def extract_business_name(doc) + def extract_title(doc) og_site_name = doc.at_css('meta[property="og:site_name"]')&.[]('content') return og_site_name.strip if og_site_name.present? @@ -47,8 +52,37 @@ class WebsiteBrandingService title&.strip&.split(/\s*[|\-–—·:]+\s*/)&.first end - def extract_language(doc) - doc.at_css('html')&.[]('lang')&.split('-')&.first&.downcase + def extract_colors(doc) + color = doc.at_css('meta[name="theme-color"]')&.[]('content') + return [] if color.blank? + + [{ hex: color, name: nil }] + end + + def extract_logos(doc) + favicon = doc.at_css('link[rel*="icon"]')&.[]('href') + return [] if favicon.blank? + + url = resolve_url(favicon) + return [] if url.blank? + + [{ url: url, type: nil, mode: nil, colors: [], resolution: { aspect_ratio: 1 } }] + end + + def build_socials(links) + handles = extract_social_from_links(links) + handles.filter_map do |platform, handle| + next if handle.blank? + + url = reconstruct_social_url(platform, handle) + { type: platform.to_s, url: url } + end + end + + def reconstruct_social_url(platform, handle) + base_urls = { whatsapp: 'https://wa.me/', line: 'https://line.me/', facebook: 'https://facebook.com/', + instagram: 'https://instagram.com/', telegram: 'https://t.me/', tiktok: 'https://tiktok.com/' } + "#{base_urls[platform]}#{handle}" end def extract_links(doc) @@ -62,24 +96,6 @@ class WebsiteBrandingService end.uniq end - def extract_branding(doc) - { - favicon: extract_favicon(doc), - primary_color: extract_theme_color(doc) - } - end - - def extract_favicon(doc) - favicon = doc.at_css('link[rel*="icon"]')&.[]('href') - return nil if favicon.blank? - - resolve_url(favicon) - end - - def extract_theme_color(doc) - doc.at_css('meta[name="theme-color"]')&.[]('content') - end - def resolve_url(url) return nil if url.blank? return url if url.start_with?('http') diff --git a/config/installation_config.yml b/config/installation_config.yml index 34cb736bf..884dd2e58 100644 --- a/config/installation_config.yml +++ b/config/installation_config.yml @@ -211,6 +211,13 @@ type: code # End of Captain Config +# ------- Context.dev Config ------- # +- name: CONTEXT_DEV_API_KEY + display_title: 'Context.dev API Key' + description: 'API key for Context.dev branding service used during account onboarding' + type: secret +# ------- End of Context.dev Config ------- # + # ------- Chatwoot Internal Config for Cloud ----# - name: CHATWOOT_INBOX_TOKEN value: diff --git a/enterprise/app/controllers/enterprise/super_admin/app_configs_controller.rb b/enterprise/app/controllers/enterprise/super_admin/app_configs_controller.rb index 87ac8f6d6..f91f12708 100644 --- a/enterprise/app/controllers/enterprise/super_admin/app_configs_controller.rb +++ b/enterprise/app/controllers/enterprise/super_admin/app_configs_controller.rb @@ -34,9 +34,9 @@ module Enterprise::SuperAdmin::AppConfigsController end def internal_config_options - %w[CHATWOOT_INBOX_TOKEN CHATWOOT_INBOX_HMAC_KEY CLOUD_ANALYTICS_TOKEN CLEARBIT_API_KEY DASHBOARD_SCRIPTS INACTIVE_WHATSAPP_NUMBERS - SKIP_INCOMING_BCC_PROCESSING CAPTAIN_CLOUD_PLAN_LIMITS ACCOUNT_SECURITY_NOTIFICATION_WEBHOOK_URL CHATWOOT_INSTANCE_ADMIN_EMAIL - OG_IMAGE_CDN_URL OG_IMAGE_CLIENT_REF CLOUDFLARE_API_KEY CLOUDFLARE_ZONE_ID BLOCKED_EMAIL_DOMAINS + %w[CHATWOOT_INBOX_TOKEN CHATWOOT_INBOX_HMAC_KEY CLOUD_ANALYTICS_TOKEN CLEARBIT_API_KEY CONTEXT_DEV_API_KEY DASHBOARD_SCRIPTS + INACTIVE_WHATSAPP_NUMBERS SKIP_INCOMING_BCC_PROCESSING CAPTAIN_CLOUD_PLAN_LIMITS ACCOUNT_SECURITY_NOTIFICATION_WEBHOOK_URL + CHATWOOT_INSTANCE_ADMIN_EMAIL OG_IMAGE_CDN_URL OG_IMAGE_CLIENT_REF CLOUDFLARE_API_KEY CLOUDFLARE_ZONE_ID BLOCKED_EMAIL_DOMAINS OTEL_PROVIDER LANGFUSE_PUBLIC_KEY LANGFUSE_SECRET_KEY LANGFUSE_BASE_URL] end diff --git a/enterprise/app/services/enterprise/website_branding_service.rb b/enterprise/app/services/enterprise/website_branding_service.rb index 6efdd5051..a1925e80d 100644 --- a/enterprise/app/services/enterprise/website_branding_service.rb +++ b/enterprise/app/services/enterprise/website_branding_service.rb @@ -1,112 +1,63 @@ module Enterprise::WebsiteBrandingService - FIRECRAWL_SCRAPE_ENDPOINT = 'https://api.firecrawl.dev/v2/scrape'.freeze - - INDUSTRY_CATEGORIES = [ - 'Technology', - 'E-commerce', - 'Healthcare', - 'Education', - 'Finance', - 'Real Estate', - 'Marketing', - 'Travel & Hospitality', - 'Food & Beverage', - 'Media & Entertainment', - 'Professional Services', - 'Non-profit', - 'Other' - ].freeze + CONTEXT_DEV_ENDPOINT = 'https://api.context.dev/v1/brand/retrieve-by-email'.freeze def perform - return super unless firecrawl_enabled? + return super unless context_dev_enabled? - response = perform_firecrawl_request - process_firecrawl_response(response) + response = fetch_brand + process_response(response) rescue StandardError => e - Rails.logger.error "[WebsiteBranding] Firecrawl failed: #{e.message}, falling back to basic scrape" - super + Rails.logger.error "[WebsiteBranding] Context.dev failed: #{e.message}" + nil end private - def firecrawl_enabled? - firecrawl_api_key.present? + def context_dev_enabled? + context_dev_api_key.present? end - def firecrawl_api_key - InstallationConfig.find_by(name: 'CAPTAIN_FIRECRAWL_API_KEY')&.value + def context_dev_api_key + InstallationConfig.find_by(name: 'CONTEXT_DEV_API_KEY')&.value end - def perform_firecrawl_request - HTTParty.post( - FIRECRAWL_SCRAPE_ENDPOINT, - body: scrape_payload.to_json, + def fetch_brand + HTTParty.get( + CONTEXT_DEV_ENDPOINT, + query: { email: @email }, headers: { - 'Authorization' => "Bearer #{firecrawl_api_key}", + 'Authorization' => "Bearer #{context_dev_api_key}", 'Content-Type' => 'application/json' } ) end - def scrape_payload - { - url: @url, - onlyMainContent: false, - formats: [ - { - type: 'json', - schema: extract_schema, - prompt: 'Extract the business name, primary language, and industry category from this website.' - }, - 'branding', - 'links' - ] - } - end - - def extract_schema - { - type: 'object', - properties: { - business_name: { type: 'string', description: 'The name of the business or company' }, - language: { type: 'string', description: 'Primary language as ISO 639-1 code (e.g., en, es, fr)' }, - industry_category: { type: 'string', enum: INDUSTRY_CATEGORIES, description: 'Industry category for this business' } - }, - required: %w[business_name] - } - end - - def process_firecrawl_response(response) + def process_response(response) + @http_status = response.code raise "API Error: #{response.message} (Status: #{response.code})" unless response.success? - format_firecrawl_response(response) + brand = response.parsed_response&.dig('brand') + return nil if brand.blank? + + format_brand(brand) end - def format_firecrawl_response(response) - data = response.parsed_response - extract = data.dig('data', 'json') || {} - brand = data.dig('data', 'branding') || {} - links = data.dig('data', 'links') || [] - + def format_brand(brand) { - business_name: extract['business_name'], - language: extract['language'], - industry_category: extract['industry_category'], - social_handles: extract_social_from_links(links), - branding: extract_firecrawl_branding(brand) - } - end - - def extract_firecrawl_branding(brand) - { - favicon: url_or_nil(brand.dig('images', 'favicon')), - primary_color: brand.dig('colors', 'primary') - } - end - - def url_or_nil(value) - return nil if value.blank? || !value.start_with?('http') - - value + domain: brand['domain'], + title: brand['title'], + description: brand['description'], + slogan: brand['slogan'], + phone: brand['phone'], + address: brand['address'], + colors: brand['colors'] || [], + logos: brand['logos'] || [], + socials: brand['socials'] || [], + links: brand['links'], + email: @email, + industries: brand.dig('industries', 'eic') || [], + stock: brand['stock'], + is_nsfw: brand['is_nsfw'] || false + }.deep_symbolize_keys end end diff --git a/lib/redis/redis_keys.rb b/lib/redis/redis_keys.rb index 8c9361ab5..59e33036d 100644 --- a/lib/redis/redis_keys.rb +++ b/lib/redis/redis_keys.rb @@ -50,6 +50,9 @@ module Redis::RedisKeys ASSIGNMENT_KEY = 'ASSIGNMENT::%d::AGENT::%d::CONVERSATION::%d'.freeze ASSIGNMENT_KEY_PATTERN = 'ASSIGNMENT::%d::AGENT::%d::*'.freeze + ## Account Onboarding + ACCOUNT_ONBOARDING_ENRICHMENT = 'ONBOARDING_ENRICHMENT::%d'.freeze + ## Account Email Rate Limiting ACCOUNT_OUTBOUND_EMAIL_COUNT_KEY = 'OUTBOUND_EMAIL_COUNT::%d::%s'.freeze end diff --git a/spec/enterprise/services/enterprise/website_branding_service_spec.rb b/spec/enterprise/services/enterprise/website_branding_service_spec.rb index 0907db518..64b6dfb53 100644 --- a/spec/enterprise/services/enterprise/website_branding_service_spec.rb +++ b/spec/enterprise/services/enterprise/website_branding_service_spec.rb @@ -7,164 +7,111 @@ end RSpec.describe Enterprise::WebsiteBrandingService do describe '#perform' do - subject(:service) { test_klass.new(url) } + subject(:service) { test_klass.new(email) } - let(:url) { 'https://example.com' } - let(:api_key) { 'test-firecrawl-api-key' } - let(:scrape_endpoint) { described_class::FIRECRAWL_SCRAPE_ENDPOINT } - let(:fallback_html) { 'Fallback' } + let(:email) { 'user@example.com' } + let(:api_key) { 'test-context-dev-api-key' } + let(:endpoint) { described_class::CONTEXT_DEV_ENDPOINT } + let(:fallback_html) { 'Fallback' } let(:success_response_body) do { - success: true, - data: { - json: { - business_name: 'Acme Corp', - language: 'en', - industry_category: 'Technology' - }, - branding: { - images: { logo: 'https://example.com/logo.png', favicon: 'https://example.com/favicon.png' }, - colors: { primary: '#FF5733' } - }, - links: [ - 'https://example.com/about', - 'https://facebook.com/acmecorp', - 'https://instagram.com/acme_corp', - 'https://wa.me/1234567890', - 'https://t.me/acmecorp', - 'https://tiktok.com/@acmetok' - ] + status: 'ok', + code: 200, + brand: { + domain: 'example.com', + title: 'Acme Corp', + description: 'Leading tech company', + slogan: 'We build things', + is_nsfw: false, + colors: [{ hex: '#FF5733', name: 'Orange Red' }], + logos: [{ url: 'https://media.brand.dev/logo.png', type: 'icon', mode: 'light', + colors: [{ hex: '#FF5733', name: 'Orange Red' }], + resolution: { width: 256, height: 256, aspect_ratio: 1 } }], + socials: [ + { type: 'facebook', url: 'https://facebook.com/acmecorp' }, + { type: 'instagram', url: 'https://instagram.com/acme_corp' } + ], + industries: { + eic: [{ industry: 'Technology', subindustry: 'Software' }] + } } }.to_json end before do - stub_request(:get, url).to_return(status: 200, body: fallback_html, headers: { 'content-type' => 'text/html' }) + stub_request(:get, 'https://example.com').to_return(status: 200, body: fallback_html, + headers: { 'content-type' => 'text/html' }) end - context 'when firecrawl is configured and API returns success' do + context 'when context.dev is configured and API returns success' do before do - create(:installation_config, name: 'CAPTAIN_FIRECRAWL_API_KEY', value: api_key) - stub_request(:post, scrape_endpoint) - .with(headers: { 'Authorization' => "Bearer #{api_key}", 'Content-Type' => 'application/json' }) + create(:installation_config, name: 'CONTEXT_DEV_API_KEY', value: api_key) + stub_request(:get, endpoint) + .with(query: { email: email }, headers: { 'Authorization' => "Bearer #{api_key}" }) .to_return(status: 200, body: success_response_body, headers: { 'content-type' => 'application/json' }) end - it 'returns business info and branding from firecrawl' do + it 'returns basic brand info' do result = service.perform - expect(result).to eq({ - business_name: 'Acme Corp', - language: 'en', - industry_category: 'Technology', - social_handles: { - whatsapp: '1234567890', - line: nil, - facebook: 'acmecorp', - instagram: 'acme_corp', - telegram: 'acmecorp', - tiktok: '@acmetok' - }, - branding: { - favicon: 'https://example.com/favicon.png', - primary_color: '#FF5733' - } - }) + expect(result).to include(domain: 'example.com', title: 'Acme Corp', description: 'Leading tech company', + slogan: 'We build things', is_nsfw: false, email: email) + end + + it 'returns colors, logos, socials, and industries' do + result = service.perform + + expect(result[:colors]).to eq([{ hex: '#FF5733', name: 'Orange Red' }]) + expect(result[:logos].first[:url]).to eq('https://media.brand.dev/logo.png') + expect(result[:socials]).to eq([{ type: 'facebook', url: 'https://facebook.com/acmecorp' }, + { type: 'instagram', url: 'https://instagram.com/acme_corp' }]) + expect(result[:industries]).to eq([{ industry: 'Technology', subindustry: 'Software' }]) end end - context 'when firecrawl API returns an error' do + context 'when context.dev API returns an error' do before do - create(:installation_config, name: 'CAPTAIN_FIRECRAWL_API_KEY', value: api_key) - stub_request(:post, scrape_endpoint) - .to_return(status: 422, body: '{"error": "Invalid URL"}', headers: {}) + create(:installation_config, name: 'CONTEXT_DEV_API_KEY', value: api_key) + stub_request(:get, endpoint) + .with(query: { email: email }) + .to_return(status: 422, body: '{"error": "FREE_EMAIL_DETECTED"}') end - it 'falls back to basic scrape' do - result = service.perform - expect(result[:business_name]).to eq('Fallback') - expect(result[:industry_category]).to be_nil + it 'returns nil' do + expect(service.perform).to be_nil end end - context 'when firecrawl raises an exception' do + context 'when context.dev raises an exception' do before do - create(:installation_config, name: 'CAPTAIN_FIRECRAWL_API_KEY', value: api_key) - stub_request(:post, scrape_endpoint).to_raise(StandardError.new('connection refused')) + create(:installation_config, name: 'CONTEXT_DEV_API_KEY', value: api_key) + stub_request(:get, endpoint).with(query: { email: email }).to_raise(StandardError.new('connection refused')) end - it 'falls back to basic scrape' do - result = service.perform - expect(result[:business_name]).to eq('Fallback') + it 'returns nil' do + expect(service.perform).to be_nil end end - context 'when firecrawl is not configured' do - it 'uses basic scrape' do - expect(HTTParty).not_to receive(:post) + context 'when context.dev is not configured' do + it 'falls back to base scraper' do result = service.perform - expect(result[:business_name]).to eq('Fallback') + expect(result[:title]).to eq('Fallback') + expect(result[:industries]).to eq([]) end end - context 'when WhatsApp link uses api.whatsapp.com format' do + context 'when context.dev returns empty brand' do before do - create(:installation_config, name: 'CAPTAIN_FIRECRAWL_API_KEY', value: api_key) - response = { - success: true, - data: { - json: { business_name: 'Acme Corp' }, - links: ['https://api.whatsapp.com/send?phone=5511999999999&text=Hello'] - } - }.to_json - stub_request(:post, scrape_endpoint) - .to_return(status: 200, body: response, headers: { 'content-type' => 'application/json' }) + create(:installation_config, name: 'CONTEXT_DEV_API_KEY', value: api_key) + stub_request(:get, endpoint) + .with(query: { email: email }) + .to_return(status: 200, body: { status: 'ok', code: 200, brand: nil }.to_json, + headers: { 'content-type' => 'application/json' }) end - it 'extracts phone number from query param' do - result = service.perform - expect(result[:social_handles][:whatsapp]).to eq('5511999999999') - end - end - - context 'when WhatsApp link uses wa.me format' do - before do - create(:installation_config, name: 'CAPTAIN_FIRECRAWL_API_KEY', value: api_key) - response = { - success: true, - data: { - json: { business_name: 'Acme Corp' }, - links: ['https://wa.me/+5511999999999'] - } - }.to_json - stub_request(:post, scrape_endpoint) - .to_return(status: 200, body: response, headers: { 'content-type' => 'application/json' }) - end - - it 'extracts phone number from path' do - result = service.perform - expect(result[:social_handles][:whatsapp]).to eq('5511999999999') - end - end - - context 'when links contain lookalike domains' do - before do - create(:installation_config, name: 'CAPTAIN_FIRECRAWL_API_KEY', value: api_key) - response = { - success: true, - data: { - json: { business_name: 'Acme Corp' }, - links: ['https://notfacebook.com/page', 'https://fakeinstagram.com/user'] - } - }.to_json - stub_request(:post, scrape_endpoint) - .to_return(status: 200, body: response, headers: { 'content-type' => 'application/json' }) - end - - it 'does not match lookalike domains' do - result = service.perform - expect(result[:social_handles][:facebook]).to be_nil - expect(result[:social_handles][:instagram]).to be_nil + it 'returns nil' do + expect(service.perform).to be_nil end end end diff --git a/spec/services/website_branding_service_spec.rb b/spec/services/website_branding_service_spec.rb index 19598fb59..e90da4c64 100644 --- a/spec/services/website_branding_service_spec.rb +++ b/spec/services/website_branding_service_spec.rb @@ -2,6 +2,7 @@ require 'rails_helper' RSpec.describe WebsiteBrandingService do describe '#perform' do + let(:email) { 'user@example.com' } let(:url) { 'https://example.com' } let(:html_body) do <<~HTML @@ -9,12 +10,21 @@ RSpec.describe WebsiteBrandingService do Acme Corp | Home - + + + -
Home
+
+ Facebook + Instagram +
+