Amazon Comprehend #X Amz Target=Comprehend 20171127.CreateDataset API

The #X Amz Target=Comprehend 20171127.CreateDataset API from Amazon Comprehend — 1 operation(s) for #x amz target=comprehend 20171127.createdataset.

Documentation

Specifications

Schemas & Data

Other Resources

OpenAPI Specification

amazon-comprehend-x-amz-target-comprehend-20171127-createdataset-api-openapi.yml Raw ↑
openapi: 3.0.0
info:
  version: '2017-11-27'
  x-release: v4
  title: 'Amazon Comprehend #X Amz Target=Comprehend 20171127.BatchDetectDominantLanguage #X Amz Target=Comprehend 20171127.BatchDetectDominantLanguage #X Amz Target=Comprehend 20171127.CreateDataset API'
  description: Amazon Comprehend is an Amazon Web Services service for gaining insight into the content of documents. Use these actions to determine the topics contained in your documents, the topics they discuss, the predominant sentiment expressed in them, the predominant language used, and more.
  x-logo:
    url: https://twitter.com/awscloud/profile_image?size=original
    backgroundColor: '#FFFFFF'
  termsOfService: https://aws.amazon.com/service-terms/
  contact:
    name: Mike Ralphson
    email: mike.ralphson@gmail.com
    url: https://github.com/mermade/aws2openapi
    x-twitter: PermittedSoc
  license:
    name: Apache 2.0 License
    url: http://www.apache.org/licenses/
  x-providerName: amazonaws.com
  x-serviceName: comprehend
  x-aws-signingName: comprehend
  x-origin:
  - contentType: application/json
    url: https://raw.githubusercontent.com/aws/aws-sdk-js/master/apis/comprehend-2017-11-27.normal.json
    converter:
      url: https://github.com/mermade/aws2openapi
      version: 1.0.0
    x-apisguru-driver: external
  x-apiClientRegistration:
    url: https://portal.aws.amazon.com/gp/aws/developer/registration/index.html?nc2=h_ct
  x-apisguru-categories:
  - cloud
  x-preferred: true
servers:
- url: http://comprehend.{region}.amazonaws.com
  variables:
    region:
      description: The AWS region
      enum:
      - us-east-1
      - us-east-2
      - us-west-1
      - us-west-2
      - us-gov-west-1
      - us-gov-east-1
      - ca-central-1
      - eu-north-1
      - eu-west-1
      - eu-west-2
      - eu-west-3
      - eu-central-1
      - eu-south-1
      - af-south-1
      - ap-northeast-1
      - ap-northeast-2
      - ap-northeast-3
      - ap-southeast-1
      - ap-southeast-2
      - ap-east-1
      - ap-south-1
      - sa-east-1
      - me-south-1
      default: us-east-1
  description: The Amazon Comprehend multi-region endpoint
- url: https://comprehend.{region}.amazonaws.com
  variables:
    region:
      description: The AWS region
      enum:
      - us-east-1
      - us-east-2
      - us-west-1
      - us-west-2
      - us-gov-west-1
      - us-gov-east-1
      - ca-central-1
      - eu-north-1
      - eu-west-1
      - eu-west-2
      - eu-west-3
      - eu-central-1
      - eu-south-1
      - af-south-1
      - ap-northeast-1
      - ap-northeast-2
      - ap-northeast-3
      - ap-southeast-1
      - ap-southeast-2
      - ap-east-1
      - ap-south-1
      - sa-east-1
      - me-south-1
      default: us-east-1
  description: The Amazon Comprehend multi-region endpoint
- url: http://comprehend.{region}.amazonaws.com.cn
  variables:
    region:
      description: The AWS region
      enum:
      - cn-north-1
      - cn-northwest-1
      default: cn-north-1
  description: The Amazon Comprehend endpoint for China (Beijing) and China (Ningxia)
- url: https://comprehend.{region}.amazonaws.com.cn
  variables:
    region:
      description: The AWS region
      enum:
      - cn-north-1
      - cn-northwest-1
      default: cn-north-1
  description: The Amazon Comprehend endpoint for China (Beijing) and China (Ningxia)
security:
- hmac: []
tags:
- name: '#X Amz Target=Comprehend 20171127.CreateDataset'
paths:
  /#X-Amz-Target=Comprehend_20171127.CreateDataset:
    parameters:
    - $ref: '#/components/parameters/X-Amz-Content-Sha256'
    - $ref: '#/components/parameters/X-Amz-Date'
    - $ref: '#/components/parameters/X-Amz-Algorithm'
    - $ref: '#/components/parameters/X-Amz-Credential'
    - $ref: '#/components/parameters/X-Amz-Security-Token'
    - $ref: '#/components/parameters/X-Amz-Signature'
    - $ref: '#/components/parameters/X-Amz-SignedHeaders'
    post:
      operationId: CreateDataset
      description: Creates a dataset to upload training or test data for a model associated with a flywheel. For more information about datasets, see <a href="https://docs.aws.amazon.com/comprehend/latest/dg/flywheels-about.html"> Flywheel overview</a> in the <i>Amazon Comprehend Developer Guide</i>.
      responses:
        '200':
          description: Success
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/CreateDatasetResponse'
              examples:
                CreateDataset200Example:
                  summary: Default CreateDataset 200
                  x-microcks-default: true
                  value:
                    DatasetArn: example
        '480':
          description: InvalidRequestException
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/InvalidRequestException'
              examples:
                CreateDataset480Example:
                  summary: Default CreateDataset 480
                  x-microcks-default: true
                  value: example
        '481':
          description: ResourceInUseException
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ResourceInUseException'
              examples:
                CreateDataset481Example:
                  summary: Default CreateDataset 481
                  x-microcks-default: true
                  value: example
        '482':
          description: TooManyTagsException
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/TooManyTagsException'
              examples:
                CreateDataset482Example:
                  summary: Default CreateDataset 482
                  x-microcks-default: true
                  value: example
        '483':
          description: TooManyRequestsException
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/TooManyRequestsException'
              examples:
                CreateDataset483Example:
                  summary: Default CreateDataset 483
                  x-microcks-default: true
                  value: example
        '484':
          description: ResourceLimitExceededException
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ResourceLimitExceededException'
              examples:
                CreateDataset484Example:
                  summary: Default CreateDataset 484
                  x-microcks-default: true
                  value: example
        '485':
          description: ResourceNotFoundException
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ResourceNotFoundException'
              examples:
                CreateDataset485Example:
                  summary: Default CreateDataset 485
                  x-microcks-default: true
                  value: example
        '486':
          description: InternalServerException
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/InternalServerException'
              examples:
                CreateDataset486Example:
                  summary: Default CreateDataset 486
                  x-microcks-default: true
                  value: example
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/CreateDatasetRequest'
      parameters:
      - name: X-Amz-Target
        in: header
        required: true
        schema:
          type: string
          enum:
          - Comprehend_20171127.CreateDataset
      summary: Amazon Comprehend Create Dataset
      x-microcks-operation:
        delay: 0
        dispatcher: FALLBACK
      tags:
      - '#X Amz Target=Comprehend 20171127.CreateDataset'
components:
  schemas:
    DatasetAugmentedManifestsListItem:
      type: object
      required:
      - AttributeNames
      - S3Uri
      properties:
        AttributeNames:
          allOf:
          - $ref: '#/components/schemas/AttributeNamesList'
          - description: <p>The JSON attribute that contains the annotations for your training documents. The number of attribute names that you specify depends on whether your augmented manifest file is the output of a single labeling job or a chained labeling job.</p> <p>If your file is the output of a single labeling job, specify the LabelAttributeName key that was used when the job was created in Ground Truth.</p> <p>If your file is the output of a chained labeling job, specify the LabelAttributeName key for one or more jobs in the chain. Each LabelAttributeName key provides the annotations from an individual job.</p>
        S3Uri:
          allOf:
          - $ref: '#/components/schemas/S3Uri'
          - description: The Amazon S3 location of the augmented manifest file.
        AnnotationDataS3Uri:
          allOf:
          - $ref: '#/components/schemas/S3Uri'
          - description: The S3 prefix to the annotation files that are referred in the augmented manifest file.
        SourceDocumentsS3Uri:
          allOf:
          - $ref: '#/components/schemas/S3Uri'
          - description: The S3 prefix to the source files (PDFs) that are referred to in the augmented manifest file.
        DocumentType:
          allOf:
          - $ref: '#/components/schemas/AugmentedManifestsDocumentTypeFormat'
          - description: <p>The type of augmented manifest. If you don't specify, the default is PlainTextDocument. </p> <p> <code>PLAIN_TEXT_DOCUMENT</code> A document type that represents any unicode text that is encoded in UTF-8.</p>
      description: An augmented manifest file that provides training data for your custom model. An augmented manifest file is a labeled dataset that is produced by Amazon SageMaker Ground Truth.
    CreateDatasetResponse:
      type: object
      properties:
        DatasetArn:
          allOf:
          - $ref: '#/components/schemas/ComprehendDatasetArn'
          - description: The ARN of the dataset.
    ClientRequestTokenString:
      type: string
      pattern: ^[a-zA-Z0-9-]+$
      minLength: 1
      maxLength: 64
    ResourceLimitExceededException: {}
    TooManyTagsException: {}
    ResourceNotFoundException: {}
    DatasetInputDataConfig:
      type: object
      properties:
        AugmentedManifests:
          allOf:
          - $ref: '#/components/schemas/DatasetAugmentedManifestsList'
          - description: 'A list of augmented manifest files that provide training data for your custom model. An augmented manifest file is a labeled dataset that is produced by Amazon SageMaker Ground Truth. '
        DataFormat:
          allOf:
          - $ref: '#/components/schemas/DatasetDataFormat'
          - description: '<p> <code>COMPREHEND_CSV</code>: The data format is a two-column CSV file, where the first column contains labels and the second column contains documents.</p> <p> <code>AUGMENTED_MANIFEST</code>: The data format </p>'
        DocumentClassifierInputDataConfig:
          allOf:
          - $ref: '#/components/schemas/DatasetDocumentClassifierInputDataConfig'
          - description: <p>The input properties for training a document classifier model. </p> <p>For more information on how the input file is formatted, see <a href="https://docs.aws.amazon.com/comprehend/latest/dg/prep-classifier-data.html">Preparing training data</a> in the Comprehend Developer Guide. </p>
        EntityRecognizerInputDataConfig:
          allOf:
          - $ref: '#/components/schemas/DatasetEntityRecognizerInputDataConfig'
          - description: The input properties for training an entity recognizer model.
      description: Specifies the format and location of the input data for the dataset.
    DatasetEntityRecognizerDocuments:
      type: object
      required:
      - S3Uri
      properties:
        S3Uri:
          allOf:
          - $ref: '#/components/schemas/S3Uri'
          - description: ' Specifies the Amazon S3 location where the documents for the dataset are located. '
        InputFormat:
          allOf:
          - $ref: '#/components/schemas/InputFormat'
          - description: ' Specifies how the text in an input file should be processed. This is optional, and the default is ONE_DOC_PER_LINE. ONE_DOC_PER_FILE - Each file is considered a separate document. Use this option when you are processing large documents, such as newspaper articles or scientific papers. ONE_DOC_PER_LINE - Each line in a file is considered a separate document. Use this option when you are processing many short documents, such as text messages.'
      description: Describes the documents submitted with a dataset for an entity recognizer model.
    CreateDatasetRequest:
      type: object
      required:
      - FlywheelArn
      - DatasetName
      - InputDataConfig
      title: CreateDatasetRequest
      properties:
        FlywheelArn:
          allOf:
          - $ref: '#/components/schemas/ComprehendFlywheelArn'
          - description: The Amazon Resource Number (ARN) of the flywheel of the flywheel to receive the data.
        DatasetName:
          allOf:
          - $ref: '#/components/schemas/ComprehendArnName'
          - description: Name of the dataset.
        DatasetType:
          allOf:
          - $ref: '#/components/schemas/DatasetType'
          - description: The dataset type. You can specify that the data in a dataset is for training the model or for testing the model.
        Description:
          allOf:
          - $ref: '#/components/schemas/Description'
          - description: Description of the dataset.
        InputDataConfig:
          allOf:
          - $ref: '#/components/schemas/DatasetInputDataConfig'
          - description: Information about the input data configuration. The type of input data varies based on the format of the input and whether the data is for a classifier model or an entity recognition model.
        ClientRequestToken:
          allOf:
          - $ref: '#/components/schemas/ClientRequestTokenString'
          - description: A unique identifier for the request. If you don't set the client request token, Amazon Comprehend generates one.
        Tags:
          allOf:
          - $ref: '#/components/schemas/TagList'
          - description: Tags for the dataset.
    S3Uri:
      type: string
      pattern: s3://[a-z0-9][\.\-a-z0-9]{1,61}[a-z0-9](/.*)?
      maxLength: 1024
    DatasetEntityRecognizerAnnotations:
      type: object
      required:
      - S3Uri
      properties:
        S3Uri:
          allOf:
          - $ref: '#/components/schemas/S3Uri'
          - description: ' Specifies the Amazon S3 location where the training documents for an entity recognizer are located. The URI must be in the same Region as the API endpoint that you are calling.'
      description: Describes the annotations associated with a entity recognizer.
    InputFormat:
      type: string
      enum:
      - ONE_DOC_PER_FILE
      - ONE_DOC_PER_LINE
    DatasetDocumentClassifierInputDataConfig:
      type: object
      required:
      - S3Uri
      properties:
        S3Uri:
          allOf:
          - $ref: '#/components/schemas/S3Uri'
          - description: <p>The Amazon S3 URI for the input data. The S3 bucket must be in the same Region as the API endpoint that you are calling. The URI can point to a single input file or it can provide the prefix for a collection of input files.</p> <p>For example, if you use the URI <code>S3://bucketName/prefix</code>, if the prefix is a single file, Amazon Comprehend uses that file as input. If more than one file begins with the prefix, Amazon Comprehend uses all of them as input.</p> <p>This parameter is required if you set <code>DataFormat</code> to <code>COMPREHEND_CSV</code>.</p>
        LabelDelimiter:
          allOf:
          - $ref: '#/components/schemas/LabelDelimiter'
          - description: Indicates the delimiter used to separate each label for training a multi-label classifier. The default delimiter between labels is a pipe (|). You can use a different character as a delimiter (if it's an allowed character) by specifying it under Delimiter for labels. If the training documents use a delimiter other than the default or the delimiter you specify, the labels on that line will be combined to make a single unique label, such as LABELLABELLABEL.
      description: <p>Describes the dataset input data configuration for a document classifier model.</p> <p>For more information on how the input file is formatted, see <a href="https://docs.aws.amazon.com/comprehend/latest/dg/prep-classifier-data.html">Preparing training data</a> in the Comprehend Developer Guide. </p>
    AttributeNamesListItem:
      type: string
      pattern: ^[a-zA-Z0-9](-*[a-zA-Z0-9])*
      minLength: 1
      maxLength: 63
    DatasetDataFormat:
      type: string
      enum:
      - COMPREHEND_CSV
      - AUGMENTED_MANIFEST
    AttributeNamesList:
      type: array
      items:
        $ref: '#/components/schemas/AttributeNamesListItem'
    DatasetType:
      type: string
      enum:
      - TRAIN
      - TEST
    DatasetEntityRecognizerEntityList:
      type: object
      required:
      - S3Uri
      properties:
        S3Uri:
          allOf:
          - $ref: '#/components/schemas/S3Uri'
          - description: Specifies the Amazon S3 location where the entity list is located.
      description: <p>Describes the dataset entity list for an entity recognizer model.</p> <p>For more information on how the input file is formatted, see <a href="https://docs.aws.amazon.com/comprehend/latest/dg/prep-training-data-cer.html">Preparing training data</a> in the Comprehend Developer Guide. </p>
    InvalidRequestException: {}
    DatasetAugmentedManifestsList:
      type: array
      items:
        $ref: '#/components/schemas/DatasetAugmentedManifestsListItem'
    LabelDelimiter:
      type: string
      pattern: ^[ ~!@#$%^*\-_+=|\\:;\t>?/]$
      minLength: 1
      maxLength: 1
    ComprehendFlywheelArn:
      type: string
      pattern: arn:aws(-[^:]+)?:comprehend:[a-zA-Z0-9-]*:[0-9]{12}:flywheel/[a-zA-Z0-9](-*[a-zA-Z0-9])*
      maxLength: 256
    ResourceInUseException: {}
    TagKey:
      type: string
      minLength: 1
      maxLength: 128
    DatasetEntityRecognizerInputDataConfig:
      type: object
      required:
      - Documents
      properties:
        Annotations:
          allOf:
          - $ref: '#/components/schemas/DatasetEntityRecognizerAnnotations'
          - description: The S3 location of the annotation documents for your custom entity recognizer.
        Documents:
          allOf:
          - $ref: '#/components/schemas/DatasetEntityRecognizerDocuments'
          - description: The format and location of the training documents for your custom entity recognizer.
        EntityList:
          allOf:
          - $ref: '#/components/schemas/DatasetEntityRecognizerEntityList'
          - description: The S3 location of the entity list for your custom entity recognizer.
      description: Specifies the format and location of the input data. You must provide either the <code>Annotations</code> parameter or the <code>EntityList</code> parameter.
    TooManyRequestsException: {}
    TagValue:
      type: string
      minLength: 0
      maxLength: 256
    InternalServerException: {}
    ComprehendDatasetArn:
      type: string
      pattern: arn:aws(-[^:]+)?:comprehend:[a-zA-Z0-9-]*:[0-9]{12}:flywheel/[a-zA-Z0-9](-*[a-zA-Z0-9])*/dataset/[a-zA-Z0-9](-*[a-zA-Z0-9])*
      maxLength: 256
    ComprehendArnName:
      type: string
      pattern: ^[a-zA-Z0-9](-*[a-zA-Z0-9])*$
      maxLength: 63
    Description:
      type: string
      pattern: ^([a-zA-Z0-9_])[\\a-zA-Z0-9_@#%*+=:?./!\s-]*$
      maxLength: 2048
    AugmentedManifestsDocumentTypeFormat:
      type: string
      enum:
      - PLAIN_TEXT_DOCUMENT
      - SEMI_STRUCTURED_DOCUMENT
    Tag:
      type: object
      required:
      - Key
      properties:
        Key:
          allOf:
          - $ref: '#/components/schemas/TagKey'
          - description: 'The initial part of a key-value pair that forms a tag associated with a given resource. For instance, if you want to show which resources are used by which departments, you might use “Department” as the key portion of the pair, with multiple possible values such as “sales,” “legal,” and “administration.” '
        Value:
          allOf:
          - $ref: '#/components/schemas/TagValue'
          - description: ' The second part of a key-value pair that forms a tag associated with a given resource. For instance, if you want to show which resources are used by which departments, you might use “Department” as the initial (key) portion of the pair, with a value of “sales” to indicate the sales department. '
      description: 'A key-value pair that adds as a metadata to a resource used by Amazon Comprehend. For example, a tag with the key-value pair ‘Department’:’Sales’ might be added to a resource to indicate its use by a particular department. '
    TagList:
      type: array
      items:
        $ref: '#/components/schemas/Tag'
  parameters:
    X-Amz-Date:
      name: X-Amz-Date
      in: header
      schema:
        type: string
      required: false
    X-Amz-SignedHeaders:
      name: X-Amz-SignedHeaders
      in: header
      schema:
        type: string
      required: false
    X-Amz-Credential:
      name: X-Amz-Credential
      in: header
      schema:
        type: string
      required: false
    X-Amz-Content-Sha256:
      name: X-Amz-Content-Sha256
      in: header
      schema:
        type: string
      required: false
    X-Amz-Algorithm:
      name: X-Amz-Algorithm
      in: header
      schema:
        type: string
      required: false
    X-Amz-Signature:
      name: X-Amz-Signature
      in: header
      schema:
        type: string
      required: false
    X-Amz-Security-Token:
      name: X-Amz-Security-Token
      in: header
      schema:
        type: string
      required: false
  securitySchemes:
    hmac:
      type: apiKey
      name: Authorization
      in: header
      description: Amazon Signature authorization v4
      x-amazon-apigateway-authtype: awsSigv4
externalDocs:
  description: Amazon Web Services documentation
  url: https://docs.aws.amazon.com/comprehend/
x-hasEquivalentPaths: true