Compare commits

...
Author SHA1 Message Date
Tanmay Deep Sharma b92ab626a8 update the flow to use openai pdf 2025-07-15 12:52:19 +07:00
Tanmay Deep Sharma 71f64fd00a update the rspec 2025-07-09 10:00:29 +07:00
Tanmay Deep Sharma 69b260a019 Merge remote-tracking branch 'origin/feat/support-pdf-upload-faq-gen' into feat/support-pdf-upload-faq-gen 2025-07-08 18:34:24 +07:00
Tanmay Deep Sharma 2e0bc94b2d fix embeddings issue 2025-07-08 18:33:32 +07:00
Tanmay Deep SharmaandGitHub 4896ecd033 Merge branch 'develop' into feat/support-pdf-upload-faq-gen 2025-07-08 16:07:00 +07:00
Tanmay Deep Sharma a347416667 do not catch exception 2025-07-08 15:42:07 +07:00
Tanmay Deep Sharma 99a3a2a69b add proper tests 2025-07-08 14:08:45 +07:00
Tanmay Deep Sharma 11e01f746e Merge remote-tracking branch 'origin/develop' into feat/support-pdf-upload-faq-gen 2025-07-08 11:55:13 +07:00
Tanmay Deep Sharma 6400585bd7 add retry on failure for pdf extraction 2025-07-08 11:55:07 +07:00
Tanmay Deep SharmaandGitHub 2663ad8e56 Merge branch 'develop' into feat/support-pdf-upload-faq-gen 2025-07-07 10:45:33 +05:30
Tanmay Deep Sharma 3dbb030900 update handoff prompt 2025-07-07 12:14:16 +07:00
Tanmay Deep Sharma 92fcfa1e59 fix test case for circleci 2025-07-04 00:00:09 +05:30
Tanmay Deep Sharma eb3360fcc4 Merge remote-tracking branch 'origin/feat/support-pdf-upload-faq-gen' into feat/support-pdf-upload-faq-gen 2025-07-03 21:52:37 +05:30
Tanmay Deep Sharma ac99250fac fix test case for circleci 2025-07-03 21:51:38 +05:30
Tanmay Deep Sharma c08d39d5b8 remove downgraded columns 2025-07-03 17:53:41 +05:30
Tanmay Deep Sharma eb0064ddf4 Merge remote-tracking branch 'origin/feat/support-pdf-upload-faq-gen' into feat/support-pdf-upload-faq-gen 2025-07-03 17:36:40 +05:30
Tanmay Deep Sharma 8f807aab17 refactor: fix review changes 2025-07-03 17:33:25 +05:30
Tanmay Deep Sharma 250d9d4dee refactor: review changes 2025-07-03 13:57:38 +05:30
Tanmay Deep Sharma 5762d1dd50 refactor: review changes 2025-07-03 12:24:07 +05:30
Tanmay Deep Sharma 9370d21612 Merge remote-tracking branch 'origin/develop' into feat/support-pdf-upload-faq-gen 2025-07-03 00:08:43 +05:30
Tanmay Deep Sharma 6681561864 refactor: fix the codeclimate 2025-07-03 00:07:16 +05:30
Tanmay Deep Sharma 6222f9b492 refactor: fix the codeclimate 2025-07-02 23:58:19 +05:30
Tanmay Deep Sharma dc48d44b51 Merge branch 'develop' into feat/support-pdf-upload-faq-gen 2025-07-02 23:34:09 +05:30
Tanmay Deep Sharma 3765720849 refactor: refactor PdfExtractionParserJob for better readability 2025-07-02 23:32:10 +05:30
Tanmay Deep Sharma 2f59d34e50 update the rpec fix linting 2025-07-02 22:40:20 +05:30
Tanmay Deep Sharma ec107a2d94 update document model file 2025-07-02 20:02:49 +05:30
Tanmay Deep Sharma 8e87e16486 improve the pdf upload functionality 2025-07-02 19:57:18 +05:30
Tanmay Deep Sharma 85c2fd10e4 refactor: separate frontend and backend code 2025-07-02 12:30:50 +05:30
Tanmay Deep Sharma 165a74229e refactor: separate frontend and backend code 2025-07-02 12:22:05 +05:30
Tanmay Deep Sharma a92394f446 feat: support for pdf to text extraction 2025-06-26 02:47:52 +05:30
Tanmay Deep Sharma ea3a84c930 feat: support uploading pdf in document knowledgebase 2025-06-26 02:41:19 +05:30
23 changed files with 699 additions and 28 deletions
+1 -1
View File
@@ -1024,4 +1024,4 @@ RUBY VERSION
ruby 3.4.4p34 ruby 3.4.4p34
BUNDLED WITH BUNDLED WITH
2.5.16 2.6.7
+5 -1
View File
@@ -63,7 +63,11 @@ Rails.application.routes.draw do
resources :copilot_threads, only: [:index, :create] do resources :copilot_threads, only: [:index, :create] do
resources :copilot_messages, only: [:index, :create] resources :copilot_messages, only: [:index, :create]
end end
resources :documents, only: [:index, :show, :create, :destroy] resources :documents, only: [:index, :show, :create, :destroy] do
collection do
post :upload_pdf
end
end
end end
resources :agent_bots, only: [:index, :create, :show, :update, :destroy] do resources :agent_bots, only: [:index, :create, :show, :update, :destroy] do
delete :avatar, on: :member delete :avatar, on: :member
+3 -1
View File
@@ -10,7 +10,7 @@
# #
# It's strongly recommended that you check this file into your version control system. # It's strongly recommended that you check this file into your version control system.
ActiveRecord::Schema[7.1].define(version: 2025_06_27_195529) do ActiveRecord::Schema[7.1].define(version: 2025_07_02_075600) do
# These extensions should be enabled to support this database # These extensions should be enabled to support this database
enable_extension "pg_stat_statements" enable_extension "pg_stat_statements"
enable_extension "pg_trgm" enable_extension "pg_trgm"
@@ -289,9 +289,11 @@ ActiveRecord::Schema[7.1].define(version: 2025_06_27_195529) do
t.datetime "created_at", null: false t.datetime "created_at", null: false
t.datetime "updated_at", null: false t.datetime "updated_at", null: false
t.integer "status", default: 0, null: false t.integer "status", default: 0, null: false
t.string "source_type", default: "url"
t.index ["account_id"], name: "index_captain_documents_on_account_id" t.index ["account_id"], name: "index_captain_documents_on_account_id"
t.index ["assistant_id", "external_link"], name: "index_captain_documents_on_assistant_id_and_external_link", unique: true t.index ["assistant_id", "external_link"], name: "index_captain_documents_on_assistant_id_and_external_link", unique: true
t.index ["assistant_id"], name: "index_captain_documents_on_assistant_id" t.index ["assistant_id"], name: "index_captain_documents_on_assistant_id"
t.index ["source_type"], name: "index_captain_documents_on_source_type"
t.index ["status"], name: "index_captain_documents_on_status" t.index ["status"], name: "index_captain_documents_on_status"
end end
@@ -3,9 +3,10 @@ class Api::V1::Accounts::Captain::DocumentsController < Api::V1::Accounts::BaseC
before_action -> { check_authorization(Captain::Assistant) } before_action -> { check_authorization(Captain::Assistant) }
before_action :set_current_page, only: [:index] before_action :set_current_page, only: [:index]
before_action :set_documents, except: [:create] before_action :set_documents, except: [:create, :upload_pdf]
before_action :set_document, only: [:show, :destroy] before_action :set_document, only: [:show, :destroy]
before_action :set_assistant, only: [:create] before_action :set_assistant, only: [:create, :upload_pdf]
RESULTS_PER_PAGE = 25 RESULTS_PER_PAGE = 25
def index def index
@@ -27,6 +28,22 @@ class Api::V1::Accounts::Captain::DocumentsController < Api::V1::Accounts::BaseC
render_could_not_create_error(e.message) render_could_not_create_error(e.message)
end end
def upload_pdf
return render_could_not_create_error('Missing Assistant') if @assistant.nil?
return render_could_not_create_error('No PDF file provided') if pdf_params[:pdf_document].blank?
@document = @assistant.documents.build(
name: pdf_params[:pdf_document].original_filename,
external_link: "pdf_upload_#{SecureRandom.hex(8)}.pdf"
)
@document.file.attach(pdf_params[:pdf_document])
@document.save!
render :create
rescue Captain::Document::LimitExceededError => e
render_could_not_create_error(e.message)
end
def destroy def destroy
@document.destroy @document.destroy
head :no_content head :no_content
@@ -43,7 +60,8 @@ class Api::V1::Accounts::Captain::DocumentsController < Api::V1::Accounts::BaseC
end end
def set_assistant def set_assistant
@assistant = Current.account.captain_assistants.find_by(id: document_params[:assistant_id]) assistant_id = action_name == 'upload_pdf' ? pdf_params[:assistant_id] : document_params[:assistant_id]
@assistant = Current.account.captain_assistants.find_by(id: assistant_id)
end end
def set_current_page def set_current_page
@@ -57,4 +75,8 @@ class Api::V1::Accounts::Captain::DocumentsController < Api::V1::Accounts::BaseC
def document_params def document_params
params.require(:document).permit(:name, :external_link, :assistant_id) params.require(:document).permit(:name, :external_link, :assistant_id)
end end
end
def pdf_params
params.permit(:pdf_document, :assistant_id)
end
end
@@ -52,12 +52,15 @@ class Captain::Conversation::ResponseBuilderJob < ApplicationJob
def message_content(message) def message_content(message)
return message.content if message.content.present? return message.content if message.content.present?
handle_message_without_content(message)
end
def handle_message_without_content(message)
return 'User has shared a message without content' unless message.attachments.any? return 'User has shared a message without content' unless message.attachments.any?
audio_transcriptions = extract_audio_transcriptions(message.attachments) audio_transcriptions = extract_audio_transcriptions(message.attachments)
return audio_transcriptions if audio_transcriptions.present? audio_transcriptions.presence || 'User has shared an attachment'
'User has shared an attachment'
end end
def extract_audio_transcriptions(attachments) def extract_audio_transcriptions(attachments)
@@ -2,13 +2,21 @@ class Captain::Documents::CrawlJob < ApplicationJob
queue_as :low queue_as :low
def perform(document) def perform(document)
if InstallationConfig.find_by(name: 'CAPTAIN_FIRECRAWL_API_KEY')&.value.present? if pdf_document?(document)
Captain::Documents::PdfExtractionJob.perform_later(document)
elsif InstallationConfig.find_by(name: 'CAPTAIN_FIRECRAWL_API_KEY')&.value.present?
perform_firecrawl_crawl(document) perform_firecrawl_crawl(document)
else else
perform_simple_crawl(document) perform_simple_crawl(document)
end end
end end
def pdf_document?(document)
return false if document.nil?
document.pdf_document?
end
private private
include Captain::FirecrawlHelper include Captain::FirecrawlHelper
@@ -0,0 +1,82 @@
class Captain::Documents::PdfExtractionJob < ApplicationJob
retry_on ActiveStorage::FileNotFoundError, attempts: 3, wait: 2.seconds
retry_on Faraday::BadRequestError, attempts: 3, wait: 2.seconds
queue_as :low
# This job runs in parallel with S3 upload
# ActiveStorage handles S3 upload asynchronously
# We process the PDF independently
def perform(document)
return unless document.pdf_document?
initialize_document_processing(document)
result = extract_pdf_content(document)
process_extraction_result(document, result)
rescue StandardError => e
handle_extraction_failure(document, e)
end
private
def initialize_document_processing(document)
document.update(status: 'in_progress')
end
def extract_pdf_content(document)
pdf_source = document.file
# Always use OpenAI direct PDF processing
pdf_service = Captain::Tools::PdfOpenaiService.new(pdf_source)
pdf_service.perform
end
def process_extraction_result(document, result)
if result[:success] && result[:content].present?
process_pdf_content(document, result[:content])
else
handle_pdf_extraction_failure(document, result)
end
end
def process_pdf_content(document, content)
# Content is an array with single item for direct PDF processing
pdf_data = content.first || {}
metadata = pdf_data[:metadata]
if metadata && metadata[:processing_type] == 'direct_pdf'
process_direct_pdf_content(document, metadata)
else
document.update!(
content: pdf_data[:content] || '',
status: 'available'
)
end
end
def process_direct_pdf_content(document, metadata)
# Store the OpenAI file_id in the document
document.update!(
content: metadata[:openai_file_id],
status: 'available'
)
# Process the PDF directly
document.responses.destroy_all
Captain::Documents::ResponseBuilderJob.perform_later(
document,
metadata[:openai_file_id],
skip_reset: false,
metadata: metadata
)
end
def handle_pdf_extraction_failure(document, _result)
document.update(status: 'available')
end
def handle_extraction_failure(document, _error)
document.update(status: 'available')
end
end
@@ -1,10 +1,28 @@
class Captain::Documents::ResponseBuilderJob < ApplicationJob class Captain::Documents::ResponseBuilderJob < ApplicationJob
queue_as :low queue_as :low
def perform(document) def perform(document, full_content = nil, skip_reset: false, metadata: {})
reset_previous_responses(document) # Use full content for FAQ generation if provided (for PDFs), otherwise use document.content
content_for_faqs = full_content || document.content
# Skip processing if no content available
return if content_for_faqs.blank?
# Only reset responses if not explicitly skipped
reset_previous_responses(document) unless skip_reset
# Check if this is a direct PDF processing
faqs = if metadata[:processing_type] == 'direct_pdf' && metadata[:openai_file_id]
Captain::Llm::PdfFaqGeneratorService.new(
content_for_faqs,
is_pdf_file: true,
metadata: metadata
).generate
# NOTE: Not deleting the file_id - keeping it for future use
else
Captain::Llm::FaqGeneratorService.new(content_for_faqs).generate
end
faqs = Captain::Llm::FaqGeneratorService.new(document.content).generate
faqs.each do |faq| faqs.each do |faq|
create_response(faq, document) create_response(faq, document)
end end
@@ -24,6 +42,7 @@ class Captain::Documents::ResponseBuilderJob < ApplicationJob
documentable: document documentable: document
) )
rescue ActiveRecord::RecordInvalid => e rescue ActiveRecord::RecordInvalid => e
Rails.logger.error "Error in creating response document: #{e.message}" # Log validation errors but continue processing other FAQs
Rails.logger.info "Skipping duplicate or invalid FAQ: #{e.message}"
end end
end end
+8 -1
View File
@@ -26,10 +26,11 @@ class Captain::Document < ApplicationRecord
belongs_to :assistant, class_name: 'Captain::Assistant' belongs_to :assistant, class_name: 'Captain::Assistant'
has_many :responses, class_name: 'Captain::AssistantResponse', dependent: :destroy, as: :documentable has_many :responses, class_name: 'Captain::AssistantResponse', dependent: :destroy, as: :documentable
belongs_to :account belongs_to :account
has_one_attached :file
validates :external_link, presence: true validates :external_link, presence: true
validates :external_link, uniqueness: { scope: :assistant_id } validates :external_link, uniqueness: { scope: :assistant_id }
validates :content, length: { maximum: 200_000 } validates :content, length: { maximum: 400_000 }
before_validation :ensure_account_id before_validation :ensure_account_id
enum status: { enum status: {
@@ -47,6 +48,10 @@ class Captain::Document < ApplicationRecord
scope :for_account, ->(account_id) { where(account_id: account_id) } scope :for_account, ->(account_id) { where(account_id: account_id) }
scope :for_assistant, ->(assistant_id) { where(assistant_id: assistant_id) } scope :for_assistant, ->(assistant_id) { where(assistant_id: assistant_id) }
def pdf_document?
file.attached? || external_link&.match?(/\.pdf$/i)
end
private private
def enqueue_crawl_job def enqueue_crawl_job
@@ -57,6 +62,8 @@ class Captain::Document < ApplicationRecord
def enqueue_response_builder_job def enqueue_response_builder_job
return if status != 'available' return if status != 'available'
# Skip auto-enqueue for PDFs as they handle FAQ generation manually with full content
return if pdf_document?
Captain::Documents::ResponseBuilderJob.perform_later(self) Captain::Documents::ResponseBuilderJob.perform_later(self)
end end
@@ -0,0 +1,104 @@
module Captain::Llm::Concerns::PdfResponsesApi
extend ActiveSupport::Concern
RESPONSES_API_URL = 'https://api.openai.com'.freeze
RESPONSES_API_ENDPOINT = '/v1/responses'.freeze
API_KEY_CONFIG_NAME = 'CAPTAIN_OPEN_AI_API_KEY'.freeze
private
def process_pdf_with_responses_api
request_body = build_pdf_request_body
response = call_responses_api(request_body)
handle_api_response(response)
end
def build_pdf_request_body
{
model: @model,
input: [
build_system_input,
build_user_input_for_pdf
],
text: {
format: { type: 'json_object' }
}
}
end
def build_system_input
{
role: 'system',
content: Captain::Llm::SystemPromptsService.faq_generator
}
end
def build_user_input_for_pdf
{
role: 'user',
content: [
{ type: 'input_file', file_id: metadata[:openai_file_id] },
{ type: 'input_text', text: build_analysis_prompt }
]
}
end
def build_analysis_prompt
if metadata[:processing_instruction]
"#{pdf_analysis_prompt} #{metadata[:processing_instruction]}"
else
pdf_analysis_prompt
end
end
def pdf_analysis_prompt
'Please analyze this PDF document and generate comprehensive FAQs based on its content.'
end
def call_responses_api(request_body)
responses_api_connection.post(RESPONSES_API_ENDPOINT) do |req|
req.headers['Authorization'] = "Bearer #{api_key}"
req.headers['Content-Type'] = 'application/json'
req.body = request_body
end
end
def responses_api_connection
@responses_api_connection ||= Faraday.new(url: RESPONSES_API_URL) do |faraday|
faraday.request :json
faraday.response :json
faraday.adapter Faraday.default_adapter
end
end
def api_key
@api_key ||= InstallationConfig.find_by!(name: API_KEY_CONFIG_NAME).value
rescue ActiveRecord::RecordNotFound
raise OpenAI::Error, "API key configuration not found: #{API_KEY_CONFIG_NAME}"
end
def handle_api_response(response)
return response.body if response.status == 200
error_message = extract_error_message(response)
Rails.logger.error "OpenAI Responses API error: #{response.status} - #{error_message}"
raise OpenAI::Error, "Responses API error: #{response.status} - #{error_message}"
end
def extract_error_message(response)
body = response.body
return body unless body.is_a?(Hash)
body.dig('error', 'message') || body['error'] || body.to_s
rescue StandardError
response.body.to_s
end
def extract_content_from_responses_api(response)
output_message = Array(response['output']).first
return nil unless output_message
content_item = output_message['content']&.find { |c| c['type'] == 'output_text' }
content_item&.dig('text')
end
end
@@ -0,0 +1,34 @@
class Captain::Llm::PdfFaqGeneratorService < Captain::Llm::FaqGeneratorService
include Captain::Llm::Concerns::PdfResponsesApi
def initialize(content_or_file_id, is_pdf_file: false, metadata: {})
@is_pdf_file = is_pdf_file
@metadata = metadata
super(content_or_file_id)
end
def generate
return super unless pdf_processing?
validate_inputs!
response = process_pdf_with_responses_api
parse_response(response)
rescue ArgumentError, OpenAI::Error
[]
end
private
attr_reader :is_pdf_file, :metadata
def pdf_processing?
is_pdf_file && metadata[:openai_file_id]
end
def validate_inputs!
return unless is_pdf_file
raise ArgumentError, 'Missing file_id for PDF processing' if metadata[:openai_file_id].blank?
raise ArgumentError, 'Invalid metadata format' unless metadata.is_a?(Hash)
end
end
@@ -1,7 +1,7 @@
class Captain::Tools::FirecrawlService class Captain::Tools::FirecrawlService
def initialize def initialize
@api_key = InstallationConfig.find_by!(name: 'CAPTAIN_FIRECRAWL_API_KEY').value @api_key = InstallationConfig.find_by!(name: 'CAPTAIN_FIRECRAWL_API_KEY').value
raise 'Missing API key' if @api_key.empty? raise 'Missing API key' if @api_key.blank?
end end
def perform(url, webhook_url, crawl_limit = 10) def perform(url, webhook_url, crawl_limit = 10)
@@ -0,0 +1,117 @@
class Captain::Tools::PdfOpenaiService
include ActiveModel::Validations
class ProcessingError < StandardError; end
attr_reader :pdf_source
MAX_PDF_SIZE = 25.megabytes
OPENAI_PURPOSE = 'assistants'.freeze
API_KEY_CONFIG = 'CAPTAIN_OPEN_AI_API_KEY'.freeze
def initialize(pdf_source)
@pdf_source = pdf_source
end
def perform
return failure_response(['Invalid PDF source']) if pdf_source.blank?
file_response = upload_pdf_to_openai
return file_response unless file_response[:success]
build_success_response(file_response[:file_id])
rescue ProcessingError => e
failure_response([e.message])
rescue StandardError => e
Rails.logger.error "PDF processing error: #{e.message}"
failure_response(['An unexpected error occurred while processing the PDF'])
end
private
def openai_client
@openai_client ||= OpenAI::Client.new(
access_token: api_key,
log_errors: Rails.env.development?
)
end
def api_key
@api_key ||= InstallationConfig.find_by!(name: API_KEY_CONFIG).value
rescue ActiveRecord::RecordNotFound
raise ProcessingError, 'OpenAI API key not configured'
end
def upload_pdf_to_openai
file_content = prepare_pdf_content
return failure_response(['Could not retrieve PDF content']) unless file_content
validate_pdf_size!(file_content)
response = openai_client.files.upload(
parameters: {
file: file_content,
purpose: OPENAI_PURPOSE
}
)
{ success: true, file_id: response['id'] }
rescue OpenAI::Error => e
failure_response(["OpenAI upload failed: #{e.message}"])
end
def prepare_pdf_content
attachment = extract_attachment
return nil unless attachment&.blob
blob = attachment.blob
create_file_io(blob.download, blob.filename.to_s, blob.content_type)
rescue StandardError => e
Rails.logger.error "Failed to prepare PDF content: #{e.message}"
nil
end
def extract_attachment
return pdf_source unless pdf_source.is_a?(ActiveStorage::Attached::One)
pdf_source.attachment
end
def validate_pdf_size!(file_content)
size = file_content.size
return if size <= MAX_PDF_SIZE
raise ProcessingError, "PDF size (#{(size / 1.megabyte).round(2)}MB) exceeds maximum allowed size (#{MAX_PDF_SIZE / 1.megabyte}MB)"
end
def create_file_io(content, filename, content_type)
StringIO.new(content).tap do |io|
io.define_singleton_method(:path) { filename }
io.define_singleton_method(:content_type) { content_type }
end
end
def build_success_response(file_id)
success_response([{
content: file_id,
metadata: {
openai_file_id: file_id,
processing_type: 'direct_pdf'
}
}])
end
def success_response(content)
{
success: true,
content: content
}
end
def failure_response(errors)
{
success: false,
errors: errors
}
end
end
@@ -8,4 +8,15 @@ json.external_link resource.external_link
json.id resource.id json.id resource.id
json.name resource.name json.name resource.name
json.status resource.status json.status resource.status
json.source_type resource.source_type
json.updated_at resource.updated_at.to_i json.updated_at resource.updated_at.to_i
# Include file information for PDF uploads
if resource.file.attached?
json.file do
json.url url_for(resource.file)
json.filename resource.file.filename
json.content_type resource.file.content_type
json.size resource.file.byte_size
end
end
@@ -220,9 +220,17 @@ RSpec.describe 'Api::V1::Accounts::Captain::Documents', type: :request do
context 'with limits exceeded' do context 'with limits exceeded' do
before do before do
# Create documents first (when there are no limits)
create_list(:captain_document, 5, assistant: assistant, account: account) create_list(:captain_document, 5, assistant: assistant, account: account)
account.update_document_usage
# Now set up the limits configuration
# First delete any existing config to avoid conflicts
InstallationConfig.where(name: 'CAPTAIN_CLOUD_PLAN_LIMITS').destroy_all
create(:installation_config, name: 'CAPTAIN_CLOUD_PLAN_LIMITS', value: captain_limits.to_json) create(:installation_config, name: 'CAPTAIN_CLOUD_PLAN_LIMITS', value: captain_limits.to_json)
account.reload # Reload to ensure changes are reflected
post "/api/v1/accounts/#{account.id}/captain/documents", post "/api/v1/accounts/#{account.id}/captain/documents",
params: valid_attributes, params: valid_attributes,
headers: admin.create_new_auth_token headers: admin.create_new_auth_token
@@ -288,4 +296,69 @@ RSpec.describe 'Api::V1::Accounts::Captain::Documents', type: :request do
end end
end end
end end
describe 'POST /api/v1/accounts/:account_id/captain/documents/upload_pdf' do
let(:pdf_file) { Rack::Test::UploadedFile.new(Rails.root.join('spec/fixtures/files/sample.pdf'), 'application/pdf') }
let(:valid_pdf_params) do
{
pdf_document: pdf_file,
assistant_id: assistant.id
}
end
context 'when it is an un-authenticated user' do
before do
post "/api/v1/accounts/#{account.id}/captain/documents/upload_pdf",
params: valid_pdf_params
end
it 'returns unauthorized status' do
expect(response).to have_http_status(:unauthorized)
end
end
context 'when it is an agent' do
it 'returns unauthorized' do
post "/api/v1/accounts/#{account.id}/captain/documents/upload_pdf",
params: valid_pdf_params,
headers: agent.create_new_auth_token
expect(response).to have_http_status(:unauthorized)
end
end
context 'when it is an admin' do
context 'with valid PDF file' do
it 'uploads PDF and creates document' do
expect do
post "/api/v1/accounts/#{account.id}/captain/documents/upload_pdf",
params: valid_pdf_params,
headers: admin.create_new_auth_token
end.to change(Captain::Document, :count).by(1)
end
it 'returns success status with document data' do
post "/api/v1/accounts/#{account.id}/captain/documents/upload_pdf",
params: valid_pdf_params,
headers: admin.create_new_auth_token
expect(response).to have_http_status(:success)
expect(json_response[:id]).to be_present
expect(json_response[:name]).to be_present
end
end
context 'without PDF file' do
before do
post "/api/v1/accounts/#{account.id}/captain/documents/upload_pdf",
params: { assistant_id: assistant.id },
headers: admin.create_new_auth_token
end
it 'returns unprocessable entity status' do
expect(response).to have_http_status(:unprocessable_entity)
end
end
end
end
end end
@@ -14,7 +14,20 @@ RSpec.describe Captain::Documents::CrawlJob, type: :job do
before do before do
allow(Captain::Tools::FirecrawlService).to receive(:new).and_return(firecrawl_service) allow(Captain::Tools::FirecrawlService).to receive(:new).and_return(firecrawl_service)
allow(firecrawl_service).to receive(:perform) allow(firecrawl_service).to receive(:perform)
create(:installation_config, name: 'CAPTAIN_FIRECRAWL_API_KEY', value: 'test-key')
# Make sure we have the Firecrawl config properly set
config = InstallationConfig.find_or_create_by(name: 'CAPTAIN_FIRECRAWL_API_KEY') do |config|
config.value = 'test-key'
end
if config.value != 'test-key'
config.value = 'test-key'
config.save!
end
# Mock simple crawl service to avoid HTTP calls if it somehow gets called
simple_crawler = instance_double(Captain::Tools::SimplePageCrawlService)
allow(Captain::Tools::SimplePageCrawlService).to receive(:new).and_return(simple_crawler)
allow(simple_crawler).to receive(:page_links).and_return([])
end end
context 'with account usage limits' do context 'with account usage limits' do
@@ -73,7 +86,6 @@ RSpec.describe Captain::Documents::CrawlJob, type: :job do
before do before do
allow(Captain::Tools::SimplePageCrawlService) allow(Captain::Tools::SimplePageCrawlService)
.to receive(:new) .to receive(:new)
.with(document.external_link)
.and_return(simple_crawler) .and_return(simple_crawler)
allow(simple_crawler).to receive(:page_links).and_return(page_links) allow(simple_crawler).to receive(:page_links).and_return(page_links)
@@ -104,6 +116,32 @@ RSpec.describe Captain::Documents::CrawlJob, type: :job do
expect(simple_crawler).to receive(:page_links) expect(simple_crawler).to receive(:page_links)
described_class.perform_now(document) described_class.perform_now(document)
end end
context 'when document is a PDF' do
let(:pdf_document) { create(:captain_document, external_link: 'https://example.com/document.pdf', source_type: 'pdf_upload') }
it 'delegates to PDFExtractionJob' do
expect(Captain::Documents::PdfExtractionJob)
.to receive(:perform_later)
.with(pdf_document)
described_class.perform_now(pdf_document)
end
end
end
describe '#pdf_document?' do
let(:job) { described_class.new }
it 'detects PDF by source type' do
pdf_doc = build(:captain_document, source_type: 'pdf_upload')
expect(job.send(:pdf_document?, pdf_doc)).to be true
end
it 'returns false for non-PDF documents' do
web_doc = build(:captain_document, external_link: 'https://example.com/page.html')
expect(job.send(:pdf_document?, web_doc)).to be false
end
end end
end end
end end
+19 -3
View File
@@ -39,12 +39,19 @@ RSpec.describe Account, type: :model do
let(:assistant) { create(:captain_assistant, account: account) } let(:assistant) { create(:captain_assistant, account: account) }
before do before do
create(:installation_config, name: 'ACCOUNT_AGENTS_LIMIT', value: 20) config = InstallationConfig.find_or_create_by(name: 'ACCOUNT_AGENTS_LIMIT') do |config|
config.value = 20
end
if config.value != 20
config.value = 20
config.save!
end
end end
describe 'when captain limits are configured' do describe 'when captain limits are configured' do
before do before do
create_list(:captain_document, 3, account: account, assistant: assistant, status: :available) create_list(:captain_document, 3, account: account, assistant: assistant, status: :available)
InstallationConfig.where(name: 'CAPTAIN_CLOUD_PLAN_LIMITS').destroy_all
create(:installation_config, name: 'CAPTAIN_CLOUD_PLAN_LIMITS', value: captain_limits.to_json) create(:installation_config, name: 'CAPTAIN_CLOUD_PLAN_LIMITS', value: captain_limits.to_json)
end end
@@ -131,6 +138,7 @@ RSpec.describe Account, type: :model do
describe 'when limits are configured for an account' do describe 'when limits are configured for an account' do
before do before do
InstallationConfig.where(name: 'CAPTAIN_CLOUD_PLAN_LIMITS').destroy_all
create(:installation_config, name: 'CAPTAIN_CLOUD_PLAN_LIMITS', value: captain_limits.to_json) create(:installation_config, name: 'CAPTAIN_CLOUD_PLAN_LIMITS', value: captain_limits.to_json)
account.update(limits: { captain_documents: 5555, captain_responses: 9999 }) account.update(limits: { captain_documents: 5555, captain_responses: 9999 })
end end
@@ -175,7 +183,9 @@ RSpec.describe Account, type: :model do
it 'returns max limits from app limit if account limit and installation config is absent' do it 'returns max limits from app limit if account limit and installation config is absent' do
account.update(limits: { agents: '' }) account.update(limits: { agents: '' })
InstallationConfig.where(name: 'ACCOUNT_AGENTS_LIMIT').update(value: '') config = InstallationConfig.find_by(name: 'ACCOUNT_AGENTS_LIMIT')
config.value = '' if config
config&.save!
expect(account.usage_limits[:agents]).to eq(ChatwootApp.max_limit) expect(account.usage_limits[:agents]).to eq(ChatwootApp.max_limit)
end end
@@ -191,7 +201,13 @@ RSpec.describe Account, type: :model do
end end
before do before do
InstallationConfig.where(name: 'CHATWOOT_CLOUD_PLAN_FEATURES').first_or_create(value: plan_features) config = InstallationConfig.find_or_create_by(name: 'CHATWOOT_CLOUD_PLAN_FEATURES') do |config|
config.value = plan_features
end
if config.value != plan_features
config.value = plan_features
config.save!
end
end end
context 'when plan_name is hacker' do context 'when plan_name is hacker' do
@@ -21,7 +21,9 @@ RSpec.describe Captain::Copilot::ChatService do
end end
before do before do
create(:installation_config, name: 'CAPTAIN_OPEN_AI_API_KEY', value: 'test-key') InstallationConfig.find_or_create_by(name: 'CAPTAIN_OPEN_AI_API_KEY') do |config|
config.value = 'test-key'
end
allow(OpenAI::Client).to receive(:new).and_return(mock_openai_client) allow(OpenAI::Client).to receive(:new).and_return(mock_openai_client)
allow(mock_openai_client).to receive(:chat).and_return({ allow(mock_openai_client).to receive(:chat).and_return({
choices: [{ message: { content: '{ "content": "Hey" }' } }] choices: [{ message: { content: '{ "content": "Hey" }' } }]
@@ -8,7 +8,9 @@ RSpec.describe Captain::Llm::ConversationFaqService do
let(:embedding_service) { instance_double(Captain::Llm::EmbeddingService) } let(:embedding_service) { instance_double(Captain::Llm::EmbeddingService) }
before do before do
create(:installation_config) { create(:installation_config, name: 'CAPTAIN_OPEN_AI_API_KEY', value: 'test-key') } InstallationConfig.find_or_create_by(name: 'CAPTAIN_OPEN_AI_API_KEY') do |config|
config.value = 'test-key'
end
allow(OpenAI::Client).to receive(:new).and_return(client) allow(OpenAI::Client).to receive(:new).and_return(client)
allow(Captain::Llm::EmbeddingService).to receive(:new).and_return(embedding_service) allow(Captain::Llm::EmbeddingService).to receive(:new).and_return(embedding_service)
end end
@@ -7,7 +7,13 @@ RSpec.describe Captain::Tools::FirecrawlService do
let(:crawl_limit) { 15 } let(:crawl_limit) { 15 }
before do before do
create(:installation_config, name: 'CAPTAIN_FIRECRAWL_API_KEY', value: api_key) config = InstallationConfig.find_or_create_by(name: 'CAPTAIN_FIRECRAWL_API_KEY') do |config|
config.value = api_key
end
if config.value != api_key
config.value = api_key
config.save!
end
end end
describe '#initialize' do describe '#initialize' do
@@ -29,17 +35,21 @@ RSpec.describe Captain::Tools::FirecrawlService do
context 'when API key is nil' do context 'when API key is nil' do
before do before do
InstallationConfig.find_by(name: 'CAPTAIN_FIRECRAWL_API_KEY').update(value: nil) config = InstallationConfig.find_by(name: 'CAPTAIN_FIRECRAWL_API_KEY')
config.value = nil
config.save!
end end
it 'raises an error' do it 'raises an error' do
expect { described_class.new }.to raise_error(NoMethodError) expect { described_class.new }.to raise_error('Missing API key')
end end
end end
context 'when API key is empty' do context 'when API key is empty' do
before do before do
InstallationConfig.find_by(name: 'CAPTAIN_FIRECRAWL_API_KEY').update(value: '') config = InstallationConfig.find_by(name: 'CAPTAIN_FIRECRAWL_API_KEY')
config.value = ''
config.save!
end end
it 'raises an error' do it 'raises an error' do
+53
View File
@@ -0,0 +1,53 @@
%PDF-1.4
1 0 obj
<<
/Type /Catalog
/Pages 2 0 R
>>
endobj
2 0 obj
<<
/Type /Pages
/Kids [3 0 R]
/Count 1
>>
endobj
3 0 obj
<<
/Type /Page
/Parent 2 0 R
/MediaBox [0 0 612 792]
/Contents 4 0 R
>>
endobj
4 0 obj
<<
/Length 44
>>
stream
BT
/F1 12 Tf
72 720 Td
(Sample PDF for testing) Tj
ET
endstream
endobj
xref
0 5
0000000000 65535 f
0000000009 00000 n
0000000058 00000 n
0000000115 00000 n
0000000202 00000 n
trailer
<<
/Size 5
/Root 1 0 R
>>
startxref
294
%%EOF
+2
View File
@@ -0,0 +1,2 @@
This is a sample text file for testing purposes.
It contains some basic text content that can be used in tests.
+62
View File
@@ -0,0 +1,62 @@
%PDF-1.4
1 0 obj
<<
/Type /Catalog
/Pages 2 0 R
>>
endobj
2 0 obj
<<
/Type /Pages
/Kids [3 0 R]
/Count 1
>>
endobj
3 0 obj
<<
/Type /Page
/Parent 2 0 R
/MediaBox [0 0 612 792]
/Contents 4 0 R
/Resources <<
/Font <<
/F1 5 0 R
>>
>>
>>
endobj
4 0 obj
<<
/Length 54
>>
stream
BT
/F1 12 Tf
72 720 Td
(Hello World from PDF\!) Tj
ET
endstream
endobj
5 0 obj
<<
/Type /Font
/Subtype /Type1
/BaseFont /Helvetica
>>
endobj
xref
0 6
0000000000 65535 f
0000000010 00000 n
0000000060 00000 n
0000000117 00000 n
0000000254 00000 n
0000000350 00000 n
trailer
<<
/Size 6
/Root 1 0 R
>>
startxref
421
%%EOF