From 2182193a928843cd8e6588682185959dc9cf3ae1 Mon Sep 17 00:00:00 2001 From: aakashb95 Date: Wed, 10 Jun 2026 23:11:12 +0530 Subject: [PATCH] draft commit --- .../captain/chat_generation_recorder.rb | 29 ++++ enterprise/app/helpers/captain/chat_helper.rb | 31 ++--- .../documentation_support_gate.rb | 66 ++++++---- .../conversation/response_builder_job.rb | 10 +- .../captain/documentation_search_service.rb | 124 +++++++++--------- .../captain/llm/assistant_chat_service.rb | 44 ++++++- .../llm/documentation_sufficiency_service.rb | 63 +++------ .../captain/llm/system_prompts_service.rb | 8 +- .../services/captain/tools/instrumentation.rb | 16 +++ .../tools/search_documentation_service.rb | 73 ++++++++++- .../captain_documentation_answerability.md | 52 ++++++++ .../documentation_sufficiency_schema.rb | 15 +-- .../lib/captain/prompts/assistant.liquid | 2 +- .../lib/captain/tools/faq_lookup_tool.rb | 2 +- .../conversation/response_builder_job_spec.rb | 81 ++++++++---- .../lib/captain/tools/faq_lookup_tool_spec.rb | 20 ++- .../assistant/agent_runner_service_spec.rb | 2 +- .../documentation_search_service_spec.rb | 43 +++--- .../llm/assistant_chat_service_spec.rb | 37 ++++++ .../documentation_sufficiency_service_spec.rb | 43 ++++++ .../search_documentation_service_spec.rb | 59 +++++++-- 21 files changed, 573 insertions(+), 247 deletions(-) create mode 100644 enterprise/docs/captain_documentation_answerability.md create mode 100644 spec/enterprise/services/captain/llm/documentation_sufficiency_service_spec.rb diff --git a/enterprise/app/helpers/captain/chat_generation_recorder.rb b/enterprise/app/helpers/captain/chat_generation_recorder.rb index 63bcdc276..6ea035330 100644 --- a/enterprise/app/helpers/captain/chat_generation_recorder.rb +++ b/enterprise/app/helpers/captain/chat_generation_recorder.rb @@ -7,6 +7,27 @@ module Captain::ChatGenerationRecorder def record_llm_generation(chat, message) return unless valid_llm_message?(message) + if defer_llm_generation?(message) + deferred_llm_generations << [chat, message] + return + end + + record_llm_generation_span(chat, message) + end + + def flush_deferred_llm_generations + deferred_llm_generations.each do |chat, message| + record_llm_generation_span(chat, message) + end + ensure + deferred_llm_generations.clear + end + + def discard_deferred_llm_generations + deferred_llm_generations.clear + end + + def record_llm_generation_span(chat, message) # Create a generation span with model and token info for Langfuse cost calculation. # Note: span duration will be near-zero since we create and end it immediately, but token counts are what Langfuse uses for cost calculation. tracer.in_span("llm.captain.#{feature_name}.generation") do |span| @@ -17,6 +38,14 @@ module Captain::ChatGenerationRecorder Rails.logger.warn "Failed to record LLM generation: #{e.message}" end + def defer_llm_generation?(message) + !message_has_tool_calls?(message) + end + + def deferred_llm_generations + @deferred_llm_generations ||= [] + end + # Skip non-LLM messages (e.g., tool results that RubyLLM processes internally). # Check for assistant role rather than token presence - some providers/streaming modes # may not return token counts, but we still want to capture the generation for evals. diff --git a/enterprise/app/helpers/captain/chat_helper.rb b/enterprise/app/helpers/captain/chat_helper.rb index 8b8ab0f60..0f0de3a90 100644 --- a/enterprise/app/helpers/captain/chat_helper.rb +++ b/enterprise/app/helpers/captain/chat_helper.rb @@ -13,7 +13,10 @@ module Captain::ChatHelper text, attachments = Captain::OpenAiMessageBuilderService.extract_text_and_attachments(last_content) response = attachments.any? ? chat.ask(text, with: attachments) : chat.ask(text) - build_response(response) + built_response = build_response(response) + after_chat_response(built_response) + flush_deferred_llm_generations + built_response end rescue StandardError => e Rails.logger.error "#{self.class.name} Assistant: #{@assistant.id}, Error in chat completion: #{e}" @@ -57,15 +60,19 @@ module Captain::ChatHelper def handle_tool_call(tool_call) persist_thinking_message(tool_call) - start_tool_span(tool_call) + start_tool_span(tool_call) unless internally_instrumented_tool?(tool_call) (@pending_tool_calls ||= []).push(tool_call) end def handle_tool_result(result) - end_tool_span(result) + end_tool_span(result) unless internally_instrumented_tool?(@pending_tool_calls&.last) persist_tool_completion end + def internally_instrumented_tool?(tool_call) + tool_call&.name.to_s == 'search_documentation' + end + def add_messages_to_chat(chat) conversation_messages[0...-1].each do |msg| text, attachments = Captain::OpenAiMessageBuilderService.extract_text_and_attachments(msg[:content]) @@ -91,21 +98,13 @@ module Captain::ChatHelper } end - def conversation_messages - @messages.reject { |m| m[:role] == 'system' || m[:role] == :system } - end + def conversation_messages = @messages.reject { |m| m[:role] == 'system' || m[:role] == :system } - def temperature - @assistant&.config&.[]('temperature').to_f || 1 - end + def temperature = @assistant&.config&.[]('temperature').to_f || 1 - def resolved_account_id - @account&.id || @assistant&.account_id - end + def resolved_account_id = @account&.id || @assistant&.account_id - def resolved_channel_type - @conversation&.inbox&.channel_type - end + def resolved_channel_type = @conversation&.inbox&.channel_type # Ensures all LLM calls and tool executions within an agentic loop # are grouped under a single trace/session in Langfuse. @@ -123,6 +122,8 @@ module Captain::ChatHelper @agent_session_active = false unless already_active end + def after_chat_response(_response) = nil + # Must be implemented by including class to identify the feature for instrumentation. # Used for Langfuse tagging and span naming. def feature_name diff --git a/enterprise/app/jobs/captain/conversation/documentation_support_gate.rb b/enterprise/app/jobs/captain/conversation/documentation_support_gate.rb index 11ab35ee4..10a865aa3 100644 --- a/enterprise/app/jobs/captain/conversation/documentation_support_gate.rb +++ b/enterprise/app/jobs/captain/conversation/documentation_support_gate.rb @@ -1,12 +1,15 @@ module Captain::Conversation::DocumentationSupportGate private - def check_documentation_support(message_history) - return unless documentation_gate_enabled? + def check_documentation_support(message_history, chat_service: nil) + return unless documentation_support_gate_enabled? return unless customer_reply? - review = review_documentation_support(message_history, documentation_evidence(message_history)) - apply_documentation_fallback(review) + searches = documentation_searches(message_history) + return if documentation_sufficiency_checked_in_tool?(searches) + + review = documentation_support_review(message_history, searches) + apply_documentation_support_decision(review, message_history, chat_service) rescue StandardError => e ChatwootExceptionTracker.new(e, account: account).capture_exception Rails.logger.warn( @@ -15,7 +18,7 @@ module Captain::Conversation::DocumentationSupportGate ) end - def documentation_gate_enabled? + def documentation_support_gate_enabled? ActiveModel::Type::Boolean.new.cast(@assistant.config['documentation_sufficiency_gate_enabled']) end @@ -26,53 +29,58 @@ module Captain::Conversation::DocumentationSupportGate !@response['handoff_tool_called'] end - def documentation_evidence(message_history) - searches = @response['documentation_searches'].to_a - return searches if searches.present? - - [no_documentation_search(last_user_message(message_history))] - end - - def no_documentation_search(query) - { - query: query, - queries: [query], - status: 'weak', - reason: 'no_documentation_search', - matches: [] - } - end - - def review_documentation_support(message_history, evidence) + def documentation_support_review(message_history, searches) Captain::Llm::DocumentationSufficiencyService.new( assistant: @assistant, conversation: @conversation ).evaluate( message_history: message_history, - assistant_response: @response['response'], - documentation_searches: evidence + documentation_searches: searches ) end + def documentation_sufficiency_checked_in_tool?(searches) + searches.any? { |search| (search[:documentation_sufficiency] || search['documentation_sufficiency']).present? } + end + + def documentation_searches(message_history) + searches = @response['documentation_searches'].to_a + return searches if searches.present? + + [missing_documentation_search(last_user_message(message_history))] + end + + def missing_documentation_search(query) + { + query: query, + queries: [query], + matches: [] + } + end + def last_user_message(message_history) message = message_history.reverse.find { |item| (item[:role] || item['role']).to_s == 'user' } message && (message[:content] || message['content']).to_s end - def apply_documentation_fallback(review) + def apply_documentation_support_decision(review, message_history, chat_service) return unless review['decision'] == 'insufficient' + if chat_service + @response.replace(chat_service.generate_documentation_gap_response(message_history: message_history)) + else + @response['response'] = default_documentation_fallback + end + @response.merge!( - 'response' => review['fallback_response'].presence || default_documentation_fallback, 'action' => 'continue', 'action_reason' => 'missing_docs_bounded_answer', 'action_source' => 'documentation_support', - 'documentation_sufficiency_reason' => review['reason'], 'documentation_sufficiency_model' => review['model'] ) end def default_documentation_fallback - "I couldn't find enough information to answer that confidently. Would you like me to connect you with a support person?" + 'I do not have enough information to answer that. Would you like me to connect you with support?' end end diff --git a/enterprise/app/jobs/captain/conversation/response_builder_job.rb b/enterprise/app/jobs/captain/conversation/response_builder_job.rb index 3cba8aa3d..8cfd7aab6 100644 --- a/enterprise/app/jobs/captain/conversation/response_builder_job.rb +++ b/enterprise/app/jobs/captain/conversation/response_builder_job.rb @@ -38,10 +38,12 @@ class Captain::Conversation::ResponseBuilderJob < ApplicationJob chat_service = Captain::Llm::AssistantChatService.new(assistant: @assistant, conversation: @conversation) @response = chat_service.generate_response( message_history: message_history - ) - @response['documentation_searches'] = chat_service.documentation_searches - check_documentation_support(message_history) if conversation_pending? - classify_v1_response_action(message_history) if conversation_pending? + ) do |response| + response['documentation_searches'] = chat_service.documentation_searches + @response = response + check_documentation_support(message_history, chat_service: chat_service) if conversation_pending? + classify_v1_response_action(message_history) if conversation_pending? + end process_response end diff --git a/enterprise/app/services/captain/documentation_search_service.rb b/enterprise/app/services/captain/documentation_search_service.rb index 9b2935dd0..fb5d77f0d 100644 --- a/enterprise/app/services/captain/documentation_search_service.rb +++ b/enterprise/app/services/captain/documentation_search_service.rb @@ -1,9 +1,64 @@ class Captain::DocumentationSearchService - # pgvector cosine distance: lower is closer. This only marks retrieval confidence; - # final answer support is checked by Captain::Llm::DocumentationSufficiencyService. - CLOSE_MATCH_DISTANCE = 0.45 TOP_MATCHES_TO_FORMAT = 5 + class << self + def serialize(result) + result.merge(matches: result[:matches].map(&:to_h)) + end + + def metadata(result) + { + match_count: result[:matches].length, + top_semantic_distance: result[:matches].first&.semantic_distance + }.compact + end + + def format_for_tool(result, no_results_message:, documentation_sufficiency: nil) + return "#{no_results_message}\n\n#{no_results_instruction}" if result[:matches].empty? + + [documentation_sufficiency_section(documentation_sufficiency), formatted_matches(result)].flatten.compact.join("\n") + end + + private + + def formatted_matches(result) + result[:matches].first(TOP_MATCHES_TO_FORMAT).map { |match| format_match(match) } + end + + def format_match(match) + response = match.response + lines = ['', "Question: #{response.question}", "Answer: #{response.answer}"] + lines << "Source: #{response.documentable.external_link}" if response.documentable.present? && response.documentable.try(:external_link) + "#{lines.join("\n")}\n" + end + + def documentation_sufficiency_section(documentation_sufficiency) + decision = documentation_sufficiency && (documentation_sufficiency[:decision] || documentation_sufficiency['decision']) + return if decision.blank? + + if decision == 'sufficient' + [ + 'Documentation support: sufficient', + 'Instruction: Use only the retrieved documentation below to answer the user.' + ].join("\n") + else + [ + 'Documentation support: insufficient', + 'Instruction: The retrieved documentation does not answer the user question.', + 'Do not answer the factual question from these results. Ask one clarifying question if useful, or offer a handoff.' + ].join("\n") + end + end + + def no_results_instruction + [ + 'Instruction: No documentation matched this query.', + 'Do not use documentation search results to make factual claims.', + 'Ask one useful follow-up question or offer a handoff.' + ].join(' ') + end + end + def initialize(scope:, account_id: nil) @scope = scope @account_id = account_id @@ -14,68 +69,7 @@ class Captain::DocumentationSearchService { query: query, queries: [query], - matches: matches, - status: status_for(matches), - reason: reason_for(matches) + matches: matches } end - - def self.serialize(result) - result.merge(matches: result[:matches].map(&:to_h)) - end - - def self.format_for_tool(result, no_results_message:) - return "#{no_results_message}\n\n#{weak_instruction}" if result[:matches].empty? - - sections = [quality_section(result)] - sections.concat(result[:matches].first(TOP_MATCHES_TO_FORMAT).map { |match| format_match(match) }) - sections.join("\n") - end - - def self.format_match(match) - response = match.response - formatted_response = "\nQuestion: #{response.question}\nAnswer: #{response.answer}\n" - if response.documentable.present? && response.documentable.try(:external_link) - formatted_response += "Source: #{response.documentable.external_link}\n" - end - formatted_response += 'Retrieval: semantic' - formatted_response += ", semantic_distance=#{format('%.4f', match.semantic_distance)}" if match.semantic_distance.present? - "#{formatted_response}\n" - end - - def self.quality_section(result) - lines = ["Search quality: #{result[:status]}", "Search reason: #{result[:reason]}"] - lines << "Search queries: #{result[:queries].join(' | ')}" if result[:queries].to_a.size > 1 - lines << weak_instruction if result[:status] == 'weak' - lines.join("\n") - end - - def self.weak_instruction - [ - 'Instruction: The retrieved documentation is missing or weak. Do not use it to make factual claims.', - "Say you couldn't find enough information to answer confidently, ask one clarifying question if useful, or ask whether", - 'the user wants to talk to a support person.' - ].join(' ') - end - - private - - def status_for(matches) - return 'weak' if matches.empty? - - close_match?(matches.first) ? 'found' : 'weak' - end - - def reason_for(matches) - return 'no_results' if matches.empty? - - top_match = matches.first - return 'semantic_match' if close_match?(top_match) - - 'low_retrieval_confidence' - end - - def close_match?(match) - match.semantic_distance.present? && match.semantic_distance <= CLOSE_MATCH_DISTANCE - end end diff --git a/enterprise/app/services/captain/llm/assistant_chat_service.rb b/enterprise/app/services/captain/llm/assistant_chat_service.rb index 52a9fdfd1..b6aaeeb6c 100644 --- a/enterprise/app/services/captain/llm/assistant_chat_service.rb +++ b/enterprise/app/services/captain/llm/assistant_chat_service.rb @@ -23,10 +23,30 @@ class Captain::Llm::AssistantChatService < Llm::BaseAiService # # NOTE: Parameters are provided as keyword arguments to improve clarity and avoid relying on # positional ordering. - def generate_response(additional_message: nil, message_history: [], role: 'user') + def generate_response(additional_message: nil, message_history: [], role: 'user', &after_response) @messages += message_history @messages << { role: role, content: additional_message } if additional_message.present? + @after_response = after_response request_chat_completion + ensure + @after_response = nil + end + + def generate_documentation_gap_response(message_history:) + previous_messages = @messages + previous_tools = @tools + previous_after_response = @after_response + + discard_deferred_llm_generations + @messages = [system_message, documentation_gap_instruction] + message_history + @tools = [] + @after_response = nil + + request_chat_completion + ensure + @messages = previous_messages + @tools = previous_tools + @after_response = previous_after_response end private @@ -36,7 +56,9 @@ class Captain::Llm::AssistantChatService < Llm::BaseAiService Captain::Tools::SearchDocumentationService.new( @assistant, user: nil, - on_search: ->(search) { @documentation_searches << search } + on_search: ->(search) { @documentation_searches << search }, + message_history: -> { conversation_messages }, + conversation: @conversation ) ] return tools unless custom_tools_enabled? @@ -57,6 +79,20 @@ class Captain::Llm::AssistantChatService < Llm::BaseAiService } end + def documentation_gap_instruction + { + role: 'system', + content: <<~PROMPT + [Documentation Support] + The retrieved documentation was not sufficient to answer the user's latest question. + Do not answer the factual question or cite the retrieved documentation. + Respond briefly in the user's language. + Ask one clarifying question if that would help, or offer a handoff. + Return the normal JSON response. + PROMPT + } + end + def custom_tools_metadata return [] unless custom_tools_enabled? @@ -89,4 +125,8 @@ class Captain::Llm::AssistantChatService < Llm::BaseAiService def feature_name 'assistant' end + + def after_chat_response(response) + @after_response&.call(response) + end end diff --git a/enterprise/app/services/captain/llm/documentation_sufficiency_service.rb b/enterprise/app/services/captain/llm/documentation_sufficiency_service.rb index 68a6b92fe..f482ec4cb 100644 --- a/enterprise/app/services/captain/llm/documentation_sufficiency_service.rb +++ b/enterprise/app/services/captain/llm/documentation_sufficiency_service.rb @@ -1,6 +1,7 @@ class Captain::Llm::DocumentationSufficiencyService < Llm::BaseAiService include Integrations::LlmInstrumentation + MODEL = 'gpt-5.4-mini'.freeze MAX_CONTEXT_MESSAGES = 6 MAX_SEARCHES = 3 MAX_MATCHES_PER_SEARCH = 5 @@ -10,13 +11,13 @@ class Captain::Llm::DocumentationSufficiencyService < Llm::BaseAiService super() @assistant = assistant @conversation = conversation + @model = MODEL @temperature = 0.0 end - def evaluate(message_history:, assistant_response:, documentation_searches:) + def evaluate(message_history:, documentation_searches:) user_prompt = inspection_user_prompt( message_history: message_history, - assistant_response: assistant_response, documentation_searches: documentation_searches ) @@ -34,42 +35,34 @@ class Captain::Llm::DocumentationSufficiencyService < Llm::BaseAiService Rails.logger.warn( "[CAPTAIN][DocumentationSufficiency] Failed for conversation #{@conversation.display_id}: #{e.class.name}: #{e.message}" ) - { 'decision' => nil, 'reason' => nil, 'error' => e.message, 'model' => @model } + { 'decision' => nil, 'error' => e.message, 'model' => @model } end private def system_prompt <<~PROMPT - You are checking whether a customer-facing assistant response is supported by retrieved documentation. + You are checking whether retrieved documentation can answer the user's latest question. - Use only the conversation context, assistant response, and retrieved documentation search results provided. + Use only the conversation context and retrieved documentation search results provided. Do not use outside knowledge. - Return "insufficient" when the assistant makes factual claims that are not supported by the retrieved documentation. + Return "sufficient" only when the retrieved documentation directly answers the user's latest question. + Return "insufficient" when the retrieved documentation is missing, unrelated, only loosely related, or does not cover the + specific entity, product, platform, integration, account object, user intent, or constraint in the latest question. Treat prior assistant messages as claims, not evidence. They do not support the new answer by themselves. - Conversation context can support the answer only when the user explicitly provided the relevant fact, constraint, or artifact. + Conversation context can clarify the latest question, but it cannot supply missing documentation evidence. Check generic support dimensions: - same entity, product, platform, integration, or account object - same user intent, not just a nearby topic - requested constraints from the user - - specific claims in the response - evidence specificity; broad docs are not enough for specific claims - Return "sufficient" only when the documentation directly supports the response, or when the response only asks a clarifying - question, gives a safe bounded no-answer, offers handoff, or restates user-provided context without adding external claims. - A response is not a safe bounded no-answer if it includes unsupported facts, steps, recommendations, examples, or links. - A response does not answer the exact question unless the retrieved documentation supports the specific claims it makes. - If documentation is missing or weak and the response gives factual claims, advice, instructions, examples, links, - product behavior, platform behavior, or account-specific statements, return "insufficient". - - If decision is "insufficient", write fallback_response in the user's language. It should be brief, say you could not find - enough information to answer confidently, and ask whether the user wants to talk to a support person when appropriate. - If decision is not "insufficient", fallback_response must be empty. + Return only the decision. Do not write a reason or customer-facing fallback copy. PROMPT end - def inspection_user_prompt(message_history:, assistant_response:, documentation_searches:) + def inspection_user_prompt(message_history:, documentation_searches:) <<~PROMPT #{format_conversation_context(message_history)} @@ -78,23 +71,16 @@ class Captain::Llm::DocumentationSufficiencyService < Llm::BaseAiService #{format_documentation_searches(documentation_searches)} - - - #{assistant_response} - PROMPT end def format_documentation_searches(searches) searches.to_a.last(MAX_SEARCHES).map.with_index(1) do |search, index| - matches = search[:matches] || search['matches'] || [] <<~SEARCH Search #{index} - query: #{search[:query] || search['query']} - status: #{search[:status] || search['status']} - reason: #{search[:reason] || search['reason']} + query: #{value(search, :query)} matches: - #{format_documentation_matches(matches)} + #{format_documentation_matches(value(search, :matches).to_a)} SEARCH end.join("\n") end @@ -102,22 +88,21 @@ class Captain::Llm::DocumentationSufficiencyService < Llm::BaseAiService def format_documentation_matches(matches) matches.to_a.first(MAX_MATCHES_PER_SEARCH).map.with_index(1) do |match, index| <<~MATCH - #{index}. question: #{match_value(match, :question)} - answer: #{truncate_text(match_value(match, :answer))} - source: #{match_value(match, :source)} - semantic_distance: #{match_value(match, :semantic_distance)} + #{index}. question: #{value(match, :question)} + answer: #{truncate_text(value(match, :answer))} + source: #{value(match, :source)} MATCH end.join("\n") end - def match_value(match, key) = match[key] || match[key.to_s] + def value(hash, key) = hash && (hash[key] || hash[key.to_s]) def normalize_messages(message_history) message_history.filter_map do |message| - role = message[:role] || message['role'] + role = value(message, :role) next if role.blank? - { role: role.to_s, content: normalize_content(message[:content] || message['content']) } + { role: role.to_s, content: normalize_content(value(message, :content)) } end end @@ -155,13 +140,10 @@ class Captain::Llm::DocumentationSufficiencyService < Llm::BaseAiService def normalize_response(parsed, raw_content) decision = parsed['decision'].to_s - reason = parsed['reason'].to_s return invalid_response(raw_content) unless Captain::DocumentationSufficiencySchema::DECISIONS.include?(decision) { 'decision' => decision, - 'reason' => reason.presence, - 'fallback_response' => parsed['fallback_response'].to_s, 'raw_response' => raw_content, 'model' => @model } @@ -170,8 +152,6 @@ class Captain::Llm::DocumentationSufficiencyService < Llm::BaseAiService def invalid_response(raw_content) { 'decision' => nil, - 'reason' => nil, - 'fallback_response' => nil, 'raw_response' => raw_content, 'error' => 'invalid_documentation_sufficiency_response', 'model' => @model @@ -202,8 +182,7 @@ class Captain::Llm::DocumentationSufficiencyService < Llm::BaseAiService searches = documentation_searches.to_a { search_count: searches.length, - search_statuses: searches.filter_map { |search| search[:status] || search['status'] }.join(','), - search_reasons: searches.filter_map { |search| search[:reason] || search['reason'] }.join(',') + match_count: searches.sum { |search| value(search, :matches).to_a.length } } end diff --git a/enterprise/app/services/captain/llm/system_prompts_service.rb b/enterprise/app/services/captain/llm/system_prompts_service.rb index 8cf91cb12..02b1811a5 100644 --- a/enterprise/app/services/captain/llm/system_prompts_service.rb +++ b/enterprise/app/services/captain/llm/system_prompts_service.rb @@ -190,8 +190,7 @@ class Captain::Llm::SystemPromptsService [Error Handling] - If the required information is not found in the provided context, respond with an appropriate message indicating that no relevant data is available. - - If a tool response says search quality is weak, do not use that tool response to make factual claims. - Say you couldn't find enough information to answer confidently, ask one clarifying question if useful, or offer a human handoff. + - If documentation search has no results, do not make factual claims from it. Say briefly that you do not have that information, ask one clarifying question if useful, or offer a human handoff. - Avoid speculating or providing unverified information. [Available Actions] @@ -263,8 +262,9 @@ class Captain::Llm::SystemPromptsService response: '', } ``` - - If the answer is not provided in context sections, Respond to the customer and ask whether they want to talk to another support agent . If they ask to Chat with another agent, return `conversation_handoff' as the response in JSON response - - If a tool response says search quality is weak, do not use that tool response to make factual claims. Say you couldn't find enough information to answer confidently, ask one clarifying question if useful, or offer a human handoff. + - For product facts, policies, account/service behavior, or how-to questions, call `search_documentation` before saying the answer is unavailable. + - If `search_documentation` does not provide the answer, respond to the customer and ask whether they want to talk to another support agent. If they ask to chat with another agent, return `conversation_handoff' as the response in JSON response + - If documentation search has no results, do not make factual claims from it. Say briefly that you do not have that information, ask one clarifying question if useful, or offer a human handoff. #{'- You MUST provide numbered citations at the appropriate places in the text.' if config['feature_citation']} #{build_tools_section(custom_tools)} diff --git a/enterprise/app/services/captain/tools/instrumentation.rb b/enterprise/app/services/captain/tools/instrumentation.rb index 2288b239e..f66ff5e64 100644 --- a/enterprise/app/services/captain/tools/instrumentation.rb +++ b/enterprise/app/services/captain/tools/instrumentation.rb @@ -3,8 +3,24 @@ module Captain::Tools::Instrumentation include Integrations::LlmInstrumentation def execute(**args) + return super unless self.class.instrument_tool_execution? + instrument_tool_call(name, args) do super end end + + def self.prepended(base) + base.extend(ClassMethods) + end + + module ClassMethods + def instrument_tool_execution? + @instrument_tool_execution != false + end + + def skip_tool_execution_instrumentation + @instrument_tool_execution = false + end + end end diff --git a/enterprise/app/services/captain/tools/search_documentation_service.rb b/enterprise/app/services/captain/tools/search_documentation_service.rb index c33d891ff..731210bfa 100644 --- a/enterprise/app/services/captain/tools/search_documentation_service.rb +++ b/enterprise/app/services/captain/tools/search_documentation_service.rb @@ -1,7 +1,12 @@ class Captain::Tools::SearchDocumentationService < Captain::Tools::BaseTool - def initialize(assistant, user: nil, on_search: nil) + include Integrations::LlmInstrumentation + skip_tool_execution_instrumentation + + def initialize(assistant, user: nil, on_search: nil, message_history: nil, conversation: nil) super(assistant, user: user) @on_search = on_search + @message_history = message_history + @conversation = conversation end def self.name @@ -18,12 +23,70 @@ class Captain::Tools::SearchDocumentationService < Captain::Tools::BaseTool .new(account: assistant.account) .translate(query, target_language: assistant.account.locale_english_name) - result = Captain::DocumentationSearchService.new( + instrument_documentation_search(query: translated_query, original_query: query) do + format_search_result(search_documentation(translated_query)) + end + end + + private + + def instrument_documentation_search(query:, original_query:, &) + arguments = { query: query } + arguments[:original_query] = original_query if original_query != query + + instrument_tool_call('search_documentation', arguments, &) + end + + def write_search_metadata(result, documentation_sufficiency = nil) + span = OpenTelemetry::Trace.current_span + metadata = Captain::DocumentationSearchService.metadata(result) + decision = documentation_sufficiency && (documentation_sufficiency[:decision] || documentation_sufficiency['decision']) + metadata[:documentation_sufficiency] = decision if decision + metadata.each do |key, value| + span.set_attribute(format(ATTR_LANGFUSE_METADATA, key), value.to_s) + span.set_attribute(format(ATTR_LANGFUSE_OBSERVATION_METADATA, key), value.to_s) + end + rescue StandardError => e + Rails.logger.warn "#{self.class.name}: Failed to write search metadata: #{e.message}" + end + + def search_documentation(query) + Captain::DocumentationSearchService.new( scope: assistant.responses.approved, account_id: assistant.account_id - ).search(translated_query) - @on_search&.call(Captain::DocumentationSearchService.serialize(result)) + ).search(query) + end - Captain::DocumentationSearchService.format_for_tool(result, no_results_message: 'No FAQs found for the given query') + def format_search_result(result) + serialized_result = Captain::DocumentationSearchService.serialize(result) + documentation_sufficiency = evaluate_documentation_sufficiency(serialized_result) + serialized_result[:documentation_sufficiency] = documentation_sufficiency if documentation_sufficiency.present? + write_search_metadata(result, documentation_sufficiency) + @on_search&.call(serialized_result) + + Captain::DocumentationSearchService.format_for_tool( + result, + no_results_message: 'No documentation found for the given query', + documentation_sufficiency: documentation_sufficiency + ) + end + + def evaluate_documentation_sufficiency(search) + return unless documentation_sufficiency_enabled? + return { 'decision' => 'insufficient', 'model' => nil } if search[:matches].blank? + + Captain::Llm::DocumentationSufficiencyService.new( + assistant: assistant, + conversation: @conversation + ).evaluate( + message_history: @message_history.call, + documentation_searches: [search] + ) + end + + def documentation_sufficiency_enabled? + @conversation.present? && + @message_history.respond_to?(:call) && + ActiveModel::Type::Boolean.new.cast(assistant.config['documentation_sufficiency_gate_enabled']) end end diff --git a/enterprise/docs/captain_documentation_answerability.md b/enterprise/docs/captain_documentation_answerability.md new file mode 100644 index 000000000..b6d55541f --- /dev/null +++ b/enterprise/docs/captain_documentation_answerability.md @@ -0,0 +1,52 @@ +# Captain Documentation Answerability Handoff + +We investigated traces where Captain answered even though documentation search was weak, missing, or unrelated. + +## What Failed + +The main pattern was not just "no docs found". It was: + +> Captain retrieved weak or nearby docs, then answered as if they supported the answer. + +We saw three failure types: + +- The model did not call `search_documentation`. +- The tool returned loosely related docs. +- The model treated "some result exists" as enough evidence. + +## What We Tried First + +We first tried a deterministic pgvector distance threshold. + +For cosine distance in pgvector: + +- Lower is better. +- `0` means very similar. +- Higher values mean less similar. + +That failed as a product guardrail. Some useful matches had higher distances, and some bad matches shared enough words to look plausible. A single threshold would need constant tuning across accounts, languages, and writing styles. + +We also avoided keyword/stopword rules because they quickly become language-specific and account-specific. + +## Current Approach + +Search and answerability are now separate. + +After `search_documentation` retrieves docs, a small LLM check asks: + +> Do these docs answer the latest user question? + +It returns only `sufficient` or `insufficient`. It does not write customer-facing copy and does not judge the assistant's draft answer. + +The decision is added to the tool output. The final assistant generation then uses it: + +- `sufficient`: answer from the retrieved docs. +- `insufficient`: do not answer the factual question; ask a clarifying question or offer handoff. + +A post-response backstop still exists for the case where the model never called the documentation tool. + +## Next Step + +Replay weak-documentation traces against this flow and compare the `sufficient` / `insufficient` decisions with the manually reviewed golden set. + +Hybrid or keyword search should be added before the answerability check later. Better retrieval should improve the docs we pass into the same check. diff --git a/enterprise/lib/captain/documentation_sufficiency_schema.rb b/enterprise/lib/captain/documentation_sufficiency_schema.rb index c09c294a4..5cac57245 100644 --- a/enterprise/lib/captain/documentation_sufficiency_schema.rb +++ b/enterprise/lib/captain/documentation_sufficiency_schema.rb @@ -1,20 +1,7 @@ class Captain::DocumentationSufficiencySchema < RubyLLM::Schema DECISIONS = %w[sufficient insufficient].freeze - REASONS = %w[ - answers_exact_question - no_documentation_used - bounded_no_answer - missing_or_weak_evidence - wrong_entity - wrong_intent - missing_constraint - generic_evidence - unsupported_specific_claim - ].freeze string :decision, enum: DECISIONS, - description: 'Use insufficient for unsupported factual answers; use sufficient only when supported or safely bounded' - string :reason, enum: REASONS, description: 'The main reason for the decision' - string :fallback_response, description: 'If insufficient, a brief user-facing fallback in the user language; otherwise empty' + description: 'Use sufficient only when retrieved documentation directly answers the latest user question' end diff --git a/enterprise/lib/captain/prompts/assistant.liquid b/enterprise/lib/captain/prompts/assistant.liquid index 003faf5c9..568b9c828 100644 --- a/enterprise/lib/captain/prompts/assistant.liquid +++ b/enterprise/lib/captain/prompts/assistant.liquid @@ -11,7 +11,7 @@ Don't digress away from your instructions, and use all the available tools at yo # Core Rules - Do not use your own understanding or training data to provide answers. Base responses strictly on the information available through your tools and provided context. - Do not share anything outside of the context provided. -- If FAQ lookup says search quality is weak, do not use that result to make factual claims. Say you couldn't find enough information to answer confidently, ask one clarifying question if useful, or use handoff. +- If FAQ lookup has no results, do not make factual claims from it. Say briefly that you do not have that information, ask one clarifying question if useful, or use handoff. - Be concise and relevant: most of your responses should be a sentence or two, unless a more detailed explanation is necessary. - Always detect the language from the user's input and reply in the same language. - When there is ambiguity, ask clarifying questions rather than make assumptions. diff --git a/enterprise/lib/captain/tools/faq_lookup_tool.rb b/enterprise/lib/captain/tools/faq_lookup_tool.rb index bfb5eb44b..ca999c2d7 100644 --- a/enterprise/lib/captain/tools/faq_lookup_tool.rb +++ b/enterprise/lib/captain/tools/faq_lookup_tool.rb @@ -14,7 +14,7 @@ class Captain::Tools::FaqLookupTool < Captain::Tools::BasePublicTool if result[:matches].empty? log_tool_usage('no_results', { query: query }) else - log_tool_usage('found_results', { query: query, count: result[:matches].size, status: result[:status], reason: result[:reason] }) + log_tool_usage('completed', { query: query, count: result[:matches].size }) end Captain::DocumentationSearchService.format_for_tool(result, no_results_message: "No relevant FAQs found for: #{query}") diff --git a/spec/enterprise/jobs/captain/conversation/response_builder_job_spec.rb b/spec/enterprise/jobs/captain/conversation/response_builder_job_spec.rb index a878b9c52..fafc1530a 100644 --- a/spec/enterprise/jobs/captain/conversation/response_builder_job_spec.rb +++ b/spec/enterprise/jobs/captain/conversation/response_builder_job_spec.rb @@ -18,7 +18,14 @@ RSpec.describe Captain::Conversation::ResponseBuilderJob, type: :job do allow(inbox).to receive(:captain_active?).and_return(true) allow(Captain::Llm::AssistantChatService).to receive(:new).and_return(mock_llm_chat_service) - allow(mock_llm_chat_service).to receive(:generate_response).and_return({ 'response' => 'Hey, welcome to Captain Specs' }) + allow(mock_llm_chat_service).to receive(:generate_response) do |**_args, &block| + response = { 'response' => 'Hey, welcome to Captain Specs' } + block&.call(response) + response + end + allow(mock_llm_chat_service).to receive(:generate_documentation_gap_response).and_return( + { 'response' => 'I do not have enough information in the documentation. Would you like me to connect you with support?' } + ) allow(mock_llm_chat_service).to receive(:documentation_searches).and_return([]) allow(Captain::Assistant::AgentRunnerService).to receive(:new).and_return(mock_agent_runner_service) allow(mock_agent_runner_service).to receive(:generate_response).and_return({ 'response' => 'Hey, welcome to Captain V2' }) @@ -155,13 +162,11 @@ RSpec.describe Captain::Conversation::ResponseBuilderJob, type: :job do end context 'when documentation sufficiency gate is enabled' do - let(:weak_search_result) do + let(:empty_search_result) do { query: 'current plan limits', queries: ['current plan limits'], - matches: [], - status: 'weak', - reason: 'no_results' + matches: [] } end @@ -173,16 +178,21 @@ RSpec.describe Captain::Conversation::ResponseBuilderJob, type: :job do ).and_return(mock_documentation_sufficiency_service) end - it 'replaces an unsupported answer with the bounded fallback from the gate' do - allow(mock_llm_chat_service).to receive(:documentation_searches).and_return([weak_search_result]) - allow(mock_llm_chat_service).to receive(:generate_response).and_return( - { 'response' => 'Your current plan has unlimited usage.' } + it 'regenerates an unsupported answer with a documentation-gap instruction' do + allow(mock_llm_chat_service).to receive(:documentation_searches).and_return([empty_search_result]) + allow(mock_llm_chat_service).to receive(:generate_response) do |**_args, &block| + response = { 'response' => 'Your current plan has unlimited usage.' } + block&.call(response) + response + end + expect(mock_llm_chat_service).to receive(:generate_documentation_gap_response).with( + message_history: [{ content: 'Hello', role: 'user' }] + ).and_return( + { 'response' => 'I do not have enough information in the documentation. Would you like me to connect you with support?' } ) allow(mock_documentation_sufficiency_service).to receive(:evaluate).and_return( { 'decision' => 'insufficient', - 'reason' => 'missing_constraint', - 'fallback_response' => "I couldn't find enough information to answer that confidently. Would you like support?", 'model' => 'gpt-4.1' } ) @@ -190,46 +200,65 @@ RSpec.describe Captain::Conversation::ResponseBuilderJob, type: :job do described_class.perform_now(conversation, assistant) expect(conversation.messages.outgoing.last.content).to eq( - "I couldn't find enough information to answer that confidently. Would you like support?" + 'I do not have enough information in the documentation. Would you like me to connect you with support?' ) expect(account.reload.usage_limits[:captain][:responses][:consumed]).to eq(1) end it 'keeps the assistant response when the gate finds enough support' do - allow(mock_llm_chat_service).to receive(:documentation_searches).and_return([weak_search_result]) - allow(mock_llm_chat_service).to receive(:generate_response).and_return( - { 'response' => 'Billing settings show current plan usage.' } - ) + allow(mock_llm_chat_service).to receive(:documentation_searches).and_return([empty_search_result]) + allow(mock_llm_chat_service).to receive(:generate_response) do |**_args, &block| + response = { 'response' => 'Billing settings show current plan usage.' } + block&.call(response) + response + end allow(mock_documentation_sufficiency_service).to receive(:evaluate).and_return( - { 'decision' => 'sufficient', 'reason' => 'answers_exact_question', 'fallback_response' => '', 'model' => 'gpt-4.1' } + { 'decision' => 'sufficient', 'model' => 'gpt-4.1' } ) + expect(mock_llm_chat_service).not_to receive(:generate_documentation_gap_response) described_class.perform_now(conversation, assistant) expect(conversation.messages.outgoing.last.content).to eq('Billing settings show current plan usage.') end - it 'checks answers even when no documentation search was recorded' do - allow(mock_llm_chat_service).to receive(:generate_response).and_return( - { 'response' => 'Your current plan costs $99 per agent per month.' } + it 'does not run a second repair when documentation support was already checked inside the tool' do + allow(mock_llm_chat_service).to receive(:documentation_searches).and_return( + [empty_search_result.merge(documentation_sufficiency: { 'decision' => 'insufficient', 'model' => 'gpt-5.4-mini' })] ) + allow(mock_llm_chat_service).to receive(:generate_response) do |**_args, &block| + response = { 'response' => 'I do not have that information in the documentation. Would you like a handoff?' } + block&.call(response) + response + end + expect(mock_documentation_sufficiency_service).not_to receive(:evaluate) + expect(mock_llm_chat_service).not_to receive(:generate_documentation_gap_response) + + described_class.perform_now(conversation, assistant) + + expect(conversation.messages.outgoing.last.content).to eq( + 'I do not have that information in the documentation. Would you like a handoff?' + ) + end + + it 'checks answers even when no documentation search was recorded' do + allow(mock_llm_chat_service).to receive(:generate_response) do |**_args, &block| + response = { 'response' => 'Your current plan costs $99 per agent per month.' } + block&.call(response) + response + end expect(mock_documentation_sufficiency_service).to receive(:evaluate).with( message_history: [{ content: 'Hello', role: 'user' }], - assistant_response: 'Your current plan costs $99 per agent per month.', documentation_searches: [ { query: 'Hello', queries: ['Hello'], - status: 'weak', - reason: 'no_documentation_search', matches: [] } ] ).and_return( { 'decision' => 'insufficient', - 'reason' => 'unsupported_specific_claim', - 'fallback_response' => "I couldn't find enough information to answer that confidently. Would you like support?", 'model' => 'gpt-4.1' } ) @@ -237,7 +266,7 @@ RSpec.describe Captain::Conversation::ResponseBuilderJob, type: :job do described_class.perform_now(conversation, assistant) expect(conversation.messages.outgoing.last.content).to eq( - "I couldn't find enough information to answer that confidently. Would you like support?" + 'I do not have enough information in the documentation. Would you like me to connect you with support?' ) end end diff --git a/spec/enterprise/lib/captain/tools/faq_lookup_tool_spec.rb b/spec/enterprise/lib/captain/tools/faq_lookup_tool_spec.rb index da786967b..93230952a 100644 --- a/spec/enterprise/lib/captain/tools/faq_lookup_tool_spec.rb +++ b/spec/enterprise/lib/captain/tools/faq_lookup_tool_spec.rb @@ -34,13 +34,11 @@ RSpec.describe Captain::Tools::FaqLookupTool, type: :model do Captain::AssistantResponse::SearchMatch.new(response: response, semantic_distance: 0.2) end - def search_result(query:, matches:, status:, reason:) + def search_result(query:, matches:) { query: query, queries: [query], - matches: matches, - status: status, - reason: reason + matches: matches } end @@ -67,7 +65,7 @@ RSpec.describe Captain::Tools::FaqLookupTool, type: :model do before do matches = [response1, response2].map { |response| search_match(response) } allow(documentation_search_service).to receive(:search).and_return( - search_result(query: 'password reset', matches: matches, status: 'found', reason: 'semantic_match') + search_result(query: 'password reset', matches: matches) ) end @@ -78,7 +76,7 @@ RSpec.describe Captain::Tools::FaqLookupTool, type: :model do expect(result).to include('Answer: Click on forgot password link') expect(result).to include('Question: How to change email?') expect(result).to include('Answer: Go to settings and update email') - expect(documentation_searches.first[:status]).to eq('found') + expect(documentation_searches.first[:matches].first[:question]).to eq('How to reset password?') end it 'includes source link when document has external_link' do @@ -92,8 +90,8 @@ RSpec.describe Captain::Tools::FaqLookupTool, type: :model do it 'logs tool usage for search' do expect(tool).to receive(:log_tool_usage).with('searching', { query: 'password reset' }) expect(tool).to receive(:log_tool_usage).with( - 'found_results', - { query: 'password reset', count: 2, status: 'found', reason: 'semantic_match' } + 'completed', + { query: 'password reset', count: 2 } ) tool.perform(tool_context, query: 'password reset') @@ -103,14 +101,14 @@ RSpec.describe Captain::Tools::FaqLookupTool, type: :model do context 'when no FAQs found' do before do allow(documentation_search_service).to receive(:search).and_return( - search_result(query: 'nonexistent topic', matches: [], status: 'weak', reason: 'no_results') + search_result(query: 'nonexistent topic', matches: []) ) end it 'returns no results message' do result = tool.perform(tool_context, query: 'nonexistent topic') expect(result).to include('No relevant FAQs found for: nonexistent topic') - expect(result).to include('Do not use it to make factual claims') + expect(result).to include('No documentation matched this query') end it 'logs tool usage for no results' do @@ -124,7 +122,7 @@ RSpec.describe Captain::Tools::FaqLookupTool, type: :model do context 'with blank query' do it 'handles empty query' do allow(documentation_search_service).to receive(:search).and_return( - search_result(query: '', matches: [], status: 'weak', reason: 'no_results') + search_result(query: '', matches: []) ) result = tool.perform(tool_context, query: '') diff --git a/spec/enterprise/services/captain/assistant/agent_runner_service_spec.rb b/spec/enterprise/services/captain/assistant/agent_runner_service_spec.rb index f616aa30c..f54c9337a 100644 --- a/spec/enterprise/services/captain/assistant/agent_runner_service_spec.rb +++ b/spec/enterprise/services/captain/assistant/agent_runner_service_spec.rb @@ -171,7 +171,7 @@ RSpec.describe Captain::Assistant::AgentRunnerService do end it 'includes documentation searches from runner state' do - searches = [{ query: 'billing', matches: [], status: 'weak', reason: 'no_results' }] + searches = [{ query: 'billing', matches: [] }] runner_context = { state: { documentation_searches: searches } } allow(mock_result).to receive(:context).and_return(runner_context) diff --git a/spec/enterprise/services/captain/documentation_search_service_spec.rb b/spec/enterprise/services/captain/documentation_search_service_spec.rb index 002addbb1..62ac74f8e 100644 --- a/spec/enterprise/services/captain/documentation_search_service_spec.rb +++ b/spec/enterprise/services/captain/documentation_search_service_spec.rb @@ -27,63 +27,70 @@ RSpec.describe Captain::DocumentationSearchService do ) end - def search_result(matches:, status:, reason:, query: 'billing') + def search_result(matches:, query: 'billing') { query: query, queries: [query], - matches: matches, - status: status, - reason: reason + matches: matches } end describe '#search' do - it 'marks weak semantic matches as weak' do + it 'returns semantic matches without assigning retrieval quality' do query = 'How do I check limits for my current monthly plan?' - weak_match = search_match(semantic_distance: 0.9) + match = search_match(semantic_distance: 0.9) - allow(scope).to receive(:search_with_metadata).with(query, account_id: 1).and_return([weak_match]) + allow(scope).to receive(:search_with_metadata).with(query, account_id: 1).and_return([match]) result = service.search(query) - expect(result[:status]).to eq('weak') - expect(result[:reason]).to eq('low_retrieval_confidence') + expect(result[:matches]).to eq([match]) expect(result[:queries]).to eq([query]) end - it 'marks close semantic matches as found' do + it 'returns an empty match list when no documentation matches' do query = 'Where do I find billing settings?' - close_match = search_match(semantic_distance: 0.2) - allow(scope).to receive(:search_with_metadata).with(query, account_id: 1).and_return([close_match]) + allow(scope).to receive(:search_with_metadata).with(query, account_id: 1).and_return([]) result = service.search(query) - expect(result[:status]).to eq('found') - expect(result[:reason]).to eq('semantic_match') + expect(result[:matches]).to eq([]) expect(result[:queries]).to eq([query]) end end describe '.format_for_tool' do it 'adds a bounded-answer instruction when no documentation is found' do - result = search_result(query: 'unknown topic', matches: [], status: 'weak', reason: 'no_results') + result = search_result(query: 'unknown topic', matches: []) formatted_result = described_class.format_for_tool(result, no_results_message: 'No FAQs found') expect(formatted_result).to include('No FAQs found') - expect(formatted_result).to include('Do not use it to make factual claims') + expect(formatted_result).to include('No documentation matched this query') end end describe '.serialize' do it 'formats search metadata for the response-level support gate' do - result = search_result(matches: [search_match(semantic_distance: 0.2)], status: 'found', reason: 'semantic_match') + result = search_result(matches: [search_match(semantic_distance: 0.2)]) serialized_result = described_class.serialize(result) - expect(serialized_result[:status]).to eq('found') expect(serialized_result[:matches].first[:semantic_distance]).to eq(0.2) end end + + describe '.metadata' do + it 'returns compact search metadata' do + result = search_result(matches: [search_match(semantic_distance: 0.2)]) + + expect(described_class.metadata(result)).to eq( + { + match_count: 1, + top_semantic_distance: 0.2 + } + ) + end + end end diff --git a/spec/enterprise/services/captain/llm/assistant_chat_service_spec.rb b/spec/enterprise/services/captain/llm/assistant_chat_service_spec.rb index 6b2cc55c8..7808f508b 100644 --- a/spec/enterprise/services/captain/llm/assistant_chat_service_spec.rb +++ b/spec/enterprise/services/captain/llm/assistant_chat_service_spec.rb @@ -63,6 +63,43 @@ RSpec.describe Captain::Llm::AssistantChatService do expect(attributes['langfuse.observation.metadata.generation_stage']).to eq('tool_call') end + + it 'runs the response hook before the agent session span closes' do + service = described_class.new(assistant: assistant, conversation: conversation) + hook_calls = [] + + expect(service).to receive(:instrument_agent_session).and_wrap_original do |original, params, &block| + original.call(params) do + result = block.call + hook_calls << :after_hook + result + end + end + + allow(mock_chat).to receive(:ask).and_return(mock_response) + service.generate_response(message_history: [{ role: 'user', content: 'Hello' }]) do |response| + hook_calls << :hook + response + end + + expect(hook_calls).to eq(%i[hook after_hook]) + end + end + + describe '#generate_documentation_gap_response' do + it 'generates a constrained response without calling tools' do + service = described_class.new(assistant: assistant, conversation: conversation) + + expect(mock_chat).not_to receive(:with_tool) + expect(mock_chat).to receive(:with_instructions).with(a_string_including('[Documentation Support]')).and_return(mock_chat) + expect(mock_chat).to receive(:ask).with('Do your documents auto refresh?').and_return(mock_response) + + response = service.generate_documentation_gap_response( + message_history: [{ role: 'user', content: 'Do your documents auto refresh?' }] + ) + + expect(response['response']).to eq('I can see the image shows a pricing table') + end end describe 'image analysis' do diff --git a/spec/enterprise/services/captain/llm/documentation_sufficiency_service_spec.rb b/spec/enterprise/services/captain/llm/documentation_sufficiency_service_spec.rb new file mode 100644 index 000000000..1d41f876d --- /dev/null +++ b/spec/enterprise/services/captain/llm/documentation_sufficiency_service_spec.rb @@ -0,0 +1,43 @@ +require 'rails_helper' + +RSpec.describe Captain::Llm::DocumentationSufficiencyService do + let(:account) { create(:account) } + let(:assistant) { create(:captain_assistant, account: account) } + let(:conversation) { create(:conversation, account: account) } + let(:service) { described_class.new(assistant: assistant, conversation: conversation) } + let(:mock_chat) { instance_double(RubyLLM::Chat) } + let(:mock_response) do + instance_double( + RubyLLM::Message, + content: { 'decision' => 'sufficient' } + ) + end + + before do + allow(RubyLLM).to receive(:chat).and_return(mock_chat) + allow(mock_chat).to receive(:with_temperature).and_return(mock_chat) + allow(mock_chat).to receive(:with_schema).and_return(mock_chat) + allow(mock_chat).to receive(:with_instructions).and_return(mock_chat) + allow(mock_chat).to receive(:ask).and_return(mock_response) + end + + describe '#evaluate' do + it 'uses the documentation support model instead of the global Captain model' do + create(:installation_config, name: 'CAPTAIN_OPEN_AI_MODEL', value: 'gpt-5.5') + + expect(RubyLLM).to receive(:chat).with(model: 'gpt-5.4-mini').and_return(mock_chat) + + result = service.evaluate( + message_history: [{ role: 'user', content: 'Who is your mascot?' }], + documentation_searches: [ + { + query: 'mascot', + matches: [{ question: 'Who is the brand mascot?', answer: 'Robin the bird.' }] + } + ] + ) + + expect(result).to include('decision' => 'sufficient', 'model' => 'gpt-5.4-mini') + end + end +end diff --git a/spec/enterprise/services/captain/tools/search_documentation_service_spec.rb b/spec/enterprise/services/captain/tools/search_documentation_service_spec.rb index 0e1564823..a2d047b36 100644 --- a/spec/enterprise/services/captain/tools/search_documentation_service_spec.rb +++ b/spec/enterprise/services/captain/tools/search_documentation_service_spec.rb @@ -27,6 +27,7 @@ RSpec.describe Captain::Tools::SearchDocumentationService do describe '#execute' do let(:documentation_search_service) { instance_double(Captain::DocumentationSearchService) } + let(:documentation_sufficiency_service) { instance_double(Captain::Llm::DocumentationSufficiencyService) } let(:translate_query_service) { instance_double(Captain::Llm::TranslateQueryService) } let!(:response) do create( @@ -49,13 +50,11 @@ RSpec.describe Captain::Tools::SearchDocumentationService do ) end - def search_result(matches:, status:, reason:) + def search_result(matches:) { query: question, queries: [question], - matches: matches, - status: status, - reason: reason + matches: matches } end @@ -71,7 +70,7 @@ RSpec.describe Captain::Tools::SearchDocumentationService do it 'returns formatted responses for the search query' do response.update(documentable: documentable) allow(documentation_search_service).to receive(:search).with(question).and_return( - search_result(matches: [match], status: 'found', reason: 'semantic_match') + search_result(matches: [match]) ) result = service.execute(query: question) @@ -79,20 +78,62 @@ RSpec.describe Captain::Tools::SearchDocumentationService do expect(result).to include(question) expect(result).to include(answer) expect(result).to include(external_link) - expect(recorded_searches.first[:status]).to eq('found') + expect(recorded_searches.first[:matches].first[:question]).to eq(question) end end context 'when no matching responses exist' do it 'returns a bounded no-results instruction' do allow(documentation_search_service).to receive(:search).with(question).and_return( - search_result(matches: [], status: 'weak', reason: 'no_results') + search_result(matches: []) ) result = service.execute(query: question) - expect(result).to include('No FAQs found for the given query') - expect(result).to include('Do not use it to make factual claims') + expect(result).to include('No documentation found for the given query') + expect(result).to include('No documentation matched this query') + end + end + + context 'when documentation sufficiency gate is enabled' do + let(:conversation) { create(:conversation, account: assistant.account) } + let(:message_history) { [{ role: 'user', content: 'Who is your mascot?' }] } + let(:service) do + described_class.new( + assistant, + on_search: ->(search) { recorded_searches << search }, + message_history: -> { message_history }, + conversation: conversation + ) + end + + before do + assistant.update!(config: assistant.config.merge('documentation_sufficiency_gate_enabled' => true)) + allow(Captain::Llm::DocumentationSufficiencyService).to receive(:new).with( + assistant: assistant, + conversation: conversation + ).and_return(documentation_sufficiency_service) + end + + it 'returns insufficient documentation support to the final assistant generation' do + allow(documentation_search_service).to receive(:search).with(question).and_return( + search_result(matches: [match]) + ) + allow(documentation_sufficiency_service).to receive(:evaluate).with( + message_history: message_history, + documentation_searches: [ + hash_including( + query: question, + matches: [hash_including(question: question, answer: answer)] + ) + ] + ).and_return({ 'decision' => 'insufficient', 'model' => 'gpt-5.4-mini' }) + + result = service.execute(query: question) + + expect(result).to include('Documentation support: insufficient') + expect(result).to include('Do not answer the factual question from these results') + expect(recorded_searches.first[:documentation_sufficiency]).to include('decision' => 'insufficient') end end end