UpTrain Evaluation API

The Evaluation API from UpTrain — 3 operation(s) for evaluation.

OpenAPI Specification

uptrain-evaluation-api-openapi.yml Raw ↑
openapi: 3.0.1
info:
  title: UpTrain Managed Auth Evaluation API
  description: Managed HTTP API for UpTrain, the open-source (Apache-2.0) LLM evaluation platform. The API grades supplied LLM input / output / context rows against a list of named checks (context relevance, factual accuracy, response completeness, conciseness, tonality, prompt injection, hallucination and more), logs results to a named project for dashboard monitoring, and performs root cause analysis on failures. These paths correspond to the public endpoints called by the uptrain Python package's APIClient (uptrain/framework/remote.py), rooted at {server_url}/api/public. The default managed server is https://demo.uptrain.ai. Requests are authenticated with an uptrain-access-token header.
  termsOfService: https://uptrain.ai/
  contact:
    name: UpTrain
    url: https://uptrain.ai/
  license:
    name: Apache 2.0
    url: https://www.apache.org/licenses/LICENSE-2.0.html
  version: 0.7.1
servers:
- url: https://demo.uptrain.ai/api/public
  description: Default UpTrain managed evaluation service
security:
- UptrainAccessToken: []
tags:
- name: Evaluation
paths:
  /evaluate:
    post:
      operationId: evaluate
      tags:
      - Evaluation
      summary: Run evaluations on a set of LLM responses.
      description: Evaluates each row of input data against the supplied list of checks and returns the original rows enriched with per-check scores and explanations. Grading LLM calls are run server-side by UpTrain.
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/EvaluateRequest'
      responses:
        '200':
          description: Evaluation results, one object per input row.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/EvaluationResults'
        '401':
          description: Invalid or missing access token.
        '422':
          description: Invalid request payload.
  /log_and_evaluate:
    post:
      operationId: logAndEvaluate
      tags:
      - Evaluation
      summary: Log data to a project and evaluate it.
      description: Logs the supplied data under a named project and runs the requested checks, persisting results so they are visible on the UpTrain dashboard with real-time monitoring. This endpoint also backs the SDK's evaluate_experiments method, which adds experiment column names for A/B comparison of prompt or model variants.
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/LogAndEvaluateRequest'
      responses:
        '200':
          description: Evaluation results, one object per input row.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/EvaluationResults'
        '401':
          description: Invalid or missing access token.
        '422':
          description: Invalid request payload.
  /evaluation_results/{project_name}:
    get:
      operationId: downloadProjectEvalResults
      tags:
      - Evaluation
      summary: Download evaluation results for a project.
      parameters:
      - name: project_name
        in: path
        required: true
        schema:
          type: string
      responses:
        '200':
          description: All evaluation results logged under the project.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/EvaluationResults'
        '404':
          description: Project not found.
components:
  schemas:
    EvaluateRequest:
      type: object
      required:
      - data
      - checks
      properties:
        data:
          type: array
          description: List of rows to evaluate.
          items:
            $ref: '#/components/schemas/EvalRow'
        checks:
          type: array
          description: List of checks to run. Each item is a preconfigured check name (for example context_relevance, factual_accuracy, response_completeness, response_conciseness, response_relevance, valid_response, tonality, prompt_injection, jailbreak_detection) or a check object.
          items:
            type: string
          example:
          - context_relevance
          - factual_accuracy
          - response_completeness
        metadata:
          type: object
          description: Optional metadata to attach to the evaluation.
          additionalProperties: true
    EvaluationResults:
      type: array
      description: The input rows echoed back, each enriched with score_<check> and explanation_<check> fields for every requested check.
      items:
        type: object
        additionalProperties: true
    LogAndEvaluateRequest:
      type: object
      required:
      - project_name
      - data
      - checks
      properties:
        project_name:
          type: string
          description: Project to log the data and results under.
        data:
          type: array
          items:
            $ref: '#/components/schemas/EvalRow'
        checks:
          type: array
          items:
            type: string
          example:
          - context_relevance
          - factual_accuracy
        metadata:
          type: object
          additionalProperties: true
    EvalRow:
      type: object
      description: A single evaluation row. The exact keys depend on the checks requested; common keys are question, response and context.
      properties:
        question:
          type: string
          description: The user query / prompt sent to the LLM.
        response:
          type: string
          description: The LLM-generated response to grade.
        context:
          type: string
          description: Retrieved context provided to the LLM (for RAG checks).
        ground_truth:
          type: string
          description: Optional reference answer for accuracy checks.
      additionalProperties: true
  securitySchemes:
    UptrainAccessToken:
      type: apiKey
      in: header
      name: uptrain-access-token
      description: UpTrain managed-service access token. Obtained from the UpTrain dashboard and supplied on every request as the uptrain-access-token header.