All files / api/src/connectors/evaluations/argument-correctness/templates main.ts

100% Statements 14/14
25% Branches 1/4
100% Functions 1/1
100% Lines 14/14

Press n or j to go to the next uncovered block, b, p or k for the previous block.

1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44          1x 2x 2x 2x                               2x     2x     2x     2x       2x 2x 2x 2x 2x 2x  
import type {
  ArgumentCorrectnessTemplateConfig,
  ArgumentCorrectnessTemplateData,
} from '@api/connectors/evaluations/argument-correctness/types';
 
export function getArgumentCorrectnessTemplate(
  data: ArgumentCorrectnessTemplateData,
): ArgumentCorrectnessTemplateConfig {
  const systemPrompt = `You are an expert evaluator for agentic systems. Assess whether the arguments provided to each tool call are correct given the task.
 
Instructions:
- Analyze the user input, the tools called (name, description, and input), and the agent's output.
- For each tool call, decide if the arguments are correct based on the task.
- Compute an overall score as (# of correct tool calls) / (total number of tool calls).
 
Return a JSON object with fields:
{
  "score": <number between 0 and 1>,
  "reasoning": "<concise explanation of overall assessment>",
  "metadata": {
    "per_tool": [ { "name": "<tool name>", "correct": <true|false>, "explanation": "<brief why>" } ]
  }
}`;
 
  const userPrompt = `Evaluate the following:
 
Input:
${data.input ?? ''}
 
Agent Output:
${data.actual_output ?? ''}
 
Tools Called (name, description, input):
${JSON.stringify(data.tools_called ?? [], null, 2)}
 
Judge whether the arguments for each tool are correct given the input.`;
 
  return {
    systemPrompt,
    userPrompt,
    outputFormat: 'json',
  } as const;
}