remove downgraded columns
This commit is contained in:
@@ -3,11 +3,9 @@ class Captain::Tools::PdfExtractionParserJob < ApplicationJob
|
||||
|
||||
def perform(assistant_id:, pdf_content:, document_id: nil)
|
||||
assistant = Captain::Assistant.find(assistant_id)
|
||||
return unless should_process_content?(pdf_content[:content], assistant.account)
|
||||
main_document = find_main_document(document_id)
|
||||
|
||||
# Find the main document instead of creating a new one
|
||||
main_document = Captain::Document.find(document_id) if document_id
|
||||
return unless main_document
|
||||
return unless can_process_content?(pdf_content[:content], assistant.account, main_document)
|
||||
|
||||
enqueue_response_builder_job_for_chunk(main_document, pdf_content)
|
||||
rescue ActiveRecord::RecordNotFound => e
|
||||
@@ -18,6 +16,16 @@ class Captain::Tools::PdfExtractionParserJob < ApplicationJob
|
||||
|
||||
private
|
||||
|
||||
def find_main_document(document_id)
|
||||
return unless document_id
|
||||
|
||||
Captain::Document.find(document_id)
|
||||
end
|
||||
|
||||
def can_process_content?(content, account, main_document)
|
||||
main_document && should_process_content?(content, account)
|
||||
end
|
||||
|
||||
def enqueue_response_builder_job_for_chunk(main_document, pdf_content)
|
||||
# Create a context string for better AI processing that includes page info
|
||||
page_info = pdf_content[:page_number] ? " (Page #{pdf_content[:page_number]})" : ''
|
||||
@@ -29,7 +37,7 @@ class Captain::Tools::PdfExtractionParserJob < ApplicationJob
|
||||
# Use the ResponseBuilderJob with full_content parameter to generate FAQs
|
||||
# This will link all FAQs to the main document while using the chunk content for AI processing
|
||||
# Skip reset since it's already done in the main PDF extraction job
|
||||
Captain::Documents::ResponseBuilderJob.perform_later(main_document, context_content, skip_reset: true)
|
||||
Captain::Documents::ResponseBuilderJob.perform_later(main_document, context_content)
|
||||
end
|
||||
|
||||
def should_process_content?(content, account)
|
||||
|
||||
@@ -4,10 +4,7 @@
|
||||
#
|
||||
# id :bigint not null, primary key
|
||||
# content :text
|
||||
# content_type :string
|
||||
# document_type :integer default(0), not null
|
||||
# external_link :string not null
|
||||
# file_size :integer
|
||||
# name :string
|
||||
# source_type :string default("url")
|
||||
# status :integer default("in_progress"), not null
|
||||
@@ -21,8 +18,6 @@
|
||||
# index_captain_documents_on_account_id (account_id)
|
||||
# index_captain_documents_on_assistant_id (assistant_id)
|
||||
# index_captain_documents_on_assistant_id_and_external_link (assistant_id,external_link) UNIQUE
|
||||
# index_captain_documents_on_content_type (content_type)
|
||||
# index_captain_documents_on_document_type (document_type)
|
||||
# index_captain_documents_on_source_type (source_type)
|
||||
# index_captain_documents_on_status (status)
|
||||
#
|
||||
@@ -58,7 +53,7 @@ class Captain::Document < ApplicationRecord
|
||||
scope :for_assistant, ->(assistant_id) { where(assistant_id: assistant_id) }
|
||||
|
||||
def pdf_document?
|
||||
source_type == 'pdf_upload' || file.attached? || pdf_url_format?
|
||||
source_type == 'pdf_upload' || file.attached?
|
||||
end
|
||||
|
||||
private
|
||||
@@ -93,17 +88,10 @@ class Captain::Document < ApplicationRecord
|
||||
def set_default_source_type
|
||||
return if source_type.present?
|
||||
|
||||
self.source_type = if file.attached? || pdf_url_format?
|
||||
self.source_type = if file.attached?
|
||||
'pdf_upload'
|
||||
else
|
||||
'url'
|
||||
end
|
||||
end
|
||||
|
||||
def pdf_url_format?
|
||||
return false if external_link.blank?
|
||||
|
||||
url = external_link.downcase
|
||||
url.end_with?('.pdf') || url.include?('/rails/active_storage/blobs/')
|
||||
end
|
||||
end
|
||||
|
||||
Reference in New Issue
Block a user