openapi: 3.0.0 info: version: '2017-11-27' x-release: v4 title: 'Amazon Comprehend #X Amz Target=Comprehend 20171127.BatchDetectDominantLanguage #X Amz Target=Comprehend 20171127.BatchDetectDominantLanguage #X Amz Target=Comprehend 20171127.CreateDataset API' description: Amazon Comprehend is an Amazon Web Services service for gaining insight into the content of documents. Use these actions to determine the topics contained in your documents, the topics they discuss, the predominant sentiment expressed in them, the predominant language used, and more. x-logo: url: https://twitter.com/awscloud/profile_image?size=original backgroundColor: '#FFFFFF' termsOfService: https://aws.amazon.com/service-terms/ contact: name: Mike Ralphson email: mike.ralphson@gmail.com url: https://github.com/mermade/aws2openapi x-twitter: PermittedSoc license: name: Apache 2.0 License url: http://www.apache.org/licenses/ x-providerName: amazonaws.com x-serviceName: comprehend x-aws-signingName: comprehend x-origin: - contentType: application/json url: https://raw.githubusercontent.com/aws/aws-sdk-js/master/apis/comprehend-2017-11-27.normal.json converter: url: https://github.com/mermade/aws2openapi version: 1.0.0 x-apisguru-driver: external x-apiClientRegistration: url: https://portal.aws.amazon.com/gp/aws/developer/registration/index.html?nc2=h_ct x-apisguru-categories: - cloud x-preferred: true servers: - url: http://comprehend.{region}.amazonaws.com variables: region: description: The AWS region enum: - us-east-1 - us-east-2 - us-west-1 - us-west-2 - us-gov-west-1 - us-gov-east-1 - ca-central-1 - eu-north-1 - eu-west-1 - eu-west-2 - eu-west-3 - eu-central-1 - eu-south-1 - af-south-1 - ap-northeast-1 - ap-northeast-2 - ap-northeast-3 - ap-southeast-1 - ap-southeast-2 - ap-east-1 - ap-south-1 - sa-east-1 - me-south-1 default: us-east-1 description: The Amazon Comprehend multi-region endpoint - url: https://comprehend.{region}.amazonaws.com variables: region: description: The AWS region enum: - us-east-1 - us-east-2 - us-west-1 - us-west-2 - us-gov-west-1 - us-gov-east-1 - ca-central-1 - eu-north-1 - eu-west-1 - eu-west-2 - eu-west-3 - eu-central-1 - eu-south-1 - af-south-1 - ap-northeast-1 - ap-northeast-2 - ap-northeast-3 - ap-southeast-1 - ap-southeast-2 - ap-east-1 - ap-south-1 - sa-east-1 - me-south-1 default: us-east-1 description: The Amazon Comprehend multi-region endpoint - url: http://comprehend.{region}.amazonaws.com.cn variables: region: description: The AWS region enum: - cn-north-1 - cn-northwest-1 default: cn-north-1 description: The Amazon Comprehend endpoint for China (Beijing) and China (Ningxia) - url: https://comprehend.{region}.amazonaws.com.cn variables: region: description: The AWS region enum: - cn-north-1 - cn-northwest-1 default: cn-north-1 description: The Amazon Comprehend endpoint for China (Beijing) and China (Ningxia) security: - hmac: [] tags: - name: '#X Amz Target=Comprehend 20171127.CreateDataset' paths: /#X-Amz-Target=Comprehend_20171127.CreateDataset: parameters: - $ref: '#/components/parameters/X-Amz-Content-Sha256' - $ref: '#/components/parameters/X-Amz-Date' - $ref: '#/components/parameters/X-Amz-Algorithm' - $ref: '#/components/parameters/X-Amz-Credential' - $ref: '#/components/parameters/X-Amz-Security-Token' - $ref: '#/components/parameters/X-Amz-Signature' - $ref: '#/components/parameters/X-Amz-SignedHeaders' post: operationId: CreateDataset description: Creates a dataset to upload training or test data for a model associated with a flywheel. For more information about datasets, see Flywheel overview in the Amazon Comprehend Developer Guide. responses: '200': description: Success content: application/json: schema: $ref: '#/components/schemas/CreateDatasetResponse' examples: CreateDataset200Example: summary: Default CreateDataset 200 x-microcks-default: true value: DatasetArn: example '480': description: InvalidRequestException content: application/json: schema: $ref: '#/components/schemas/InvalidRequestException' examples: CreateDataset480Example: summary: Default CreateDataset 480 x-microcks-default: true value: example '481': description: ResourceInUseException content: application/json: schema: $ref: '#/components/schemas/ResourceInUseException' examples: CreateDataset481Example: summary: Default CreateDataset 481 x-microcks-default: true value: example '482': description: TooManyTagsException content: application/json: schema: $ref: '#/components/schemas/TooManyTagsException' examples: CreateDataset482Example: summary: Default CreateDataset 482 x-microcks-default: true value: example '483': description: TooManyRequestsException content: application/json: schema: $ref: '#/components/schemas/TooManyRequestsException' examples: CreateDataset483Example: summary: Default CreateDataset 483 x-microcks-default: true value: example '484': description: ResourceLimitExceededException content: application/json: schema: $ref: '#/components/schemas/ResourceLimitExceededException' examples: CreateDataset484Example: summary: Default CreateDataset 484 x-microcks-default: true value: example '485': description: ResourceNotFoundException content: application/json: schema: $ref: '#/components/schemas/ResourceNotFoundException' examples: CreateDataset485Example: summary: Default CreateDataset 485 x-microcks-default: true value: example '486': description: InternalServerException content: application/json: schema: $ref: '#/components/schemas/InternalServerException' examples: CreateDataset486Example: summary: Default CreateDataset 486 x-microcks-default: true value: example requestBody: required: true content: application/json: schema: $ref: '#/components/schemas/CreateDatasetRequest' parameters: - name: X-Amz-Target in: header required: true schema: type: string enum: - Comprehend_20171127.CreateDataset summary: Amazon Comprehend Create Dataset x-microcks-operation: delay: 0 dispatcher: FALLBACK tags: - '#X Amz Target=Comprehend 20171127.CreateDataset' components: schemas: DatasetAugmentedManifestsListItem: type: object required: - AttributeNames - S3Uri properties: AttributeNames: allOf: - $ref: '#/components/schemas/AttributeNamesList' - description:
The JSON attribute that contains the annotations for your training documents. The number of attribute names that you specify depends on whether your augmented manifest file is the output of a single labeling job or a chained labeling job.
If your file is the output of a single labeling job, specify the LabelAttributeName key that was used when the job was created in Ground Truth.
If your file is the output of a chained labeling job, specify the LabelAttributeName key for one or more jobs in the chain. Each LabelAttributeName key provides the annotations from an individual job.
S3Uri: allOf: - $ref: '#/components/schemas/S3Uri' - description: The Amazon S3 location of the augmented manifest file. AnnotationDataS3Uri: allOf: - $ref: '#/components/schemas/S3Uri' - description: The S3 prefix to the annotation files that are referred in the augmented manifest file. SourceDocumentsS3Uri: allOf: - $ref: '#/components/schemas/S3Uri' - description: The S3 prefix to the source files (PDFs) that are referred to in the augmented manifest file. DocumentType: allOf: - $ref: '#/components/schemas/AugmentedManifestsDocumentTypeFormat' - description:The type of augmented manifest. If you don't specify, the default is PlainTextDocument.
PLAIN_TEXT_DOCUMENT A document type that represents any unicode text that is encoded in UTF-8.
COMPREHEND_CSV: The data format is a two-column CSV file, where the first column contains labels and the second column contains documents.
AUGMENTED_MANIFEST: The data format
The input properties for training a document classifier model.
For more information on how the input file is formatted, see Preparing training data in the Comprehend Developer Guide.
EntityRecognizerInputDataConfig: allOf: - $ref: '#/components/schemas/DatasetEntityRecognizerInputDataConfig' - description: The input properties for training an entity recognizer model. description: Specifies the format and location of the input data for the dataset. DatasetEntityRecognizerDocuments: type: object required: - S3Uri properties: S3Uri: allOf: - $ref: '#/components/schemas/S3Uri' - description: ' Specifies the Amazon S3 location where the documents for the dataset are located. ' InputFormat: allOf: - $ref: '#/components/schemas/InputFormat' - description: ' Specifies how the text in an input file should be processed. This is optional, and the default is ONE_DOC_PER_LINE. ONE_DOC_PER_FILE - Each file is considered a separate document. Use this option when you are processing large documents, such as newspaper articles or scientific papers. ONE_DOC_PER_LINE - Each line in a file is considered a separate document. Use this option when you are processing many short documents, such as text messages.' description: Describes the documents submitted with a dataset for an entity recognizer model. CreateDatasetRequest: type: object required: - FlywheelArn - DatasetName - InputDataConfig title: CreateDatasetRequest properties: FlywheelArn: allOf: - $ref: '#/components/schemas/ComprehendFlywheelArn' - description: The Amazon Resource Number (ARN) of the flywheel of the flywheel to receive the data. DatasetName: allOf: - $ref: '#/components/schemas/ComprehendArnName' - description: Name of the dataset. DatasetType: allOf: - $ref: '#/components/schemas/DatasetType' - description: The dataset type. You can specify that the data in a dataset is for training the model or for testing the model. Description: allOf: - $ref: '#/components/schemas/Description' - description: Description of the dataset. InputDataConfig: allOf: - $ref: '#/components/schemas/DatasetInputDataConfig' - description: Information about the input data configuration. The type of input data varies based on the format of the input and whether the data is for a classifier model or an entity recognition model. ClientRequestToken: allOf: - $ref: '#/components/schemas/ClientRequestTokenString' - description: A unique identifier for the request. If you don't set the client request token, Amazon Comprehend generates one. Tags: allOf: - $ref: '#/components/schemas/TagList' - description: Tags for the dataset. S3Uri: type: string pattern: s3://[a-z0-9][\.\-a-z0-9]{1,61}[a-z0-9](/.*)? maxLength: 1024 DatasetEntityRecognizerAnnotations: type: object required: - S3Uri properties: S3Uri: allOf: - $ref: '#/components/schemas/S3Uri' - description: ' Specifies the Amazon S3 location where the training documents for an entity recognizer are located. The URI must be in the same Region as the API endpoint that you are calling.' description: Describes the annotations associated with a entity recognizer. InputFormat: type: string enum: - ONE_DOC_PER_FILE - ONE_DOC_PER_LINE DatasetDocumentClassifierInputDataConfig: type: object required: - S3Uri properties: S3Uri: allOf: - $ref: '#/components/schemas/S3Uri' - description:The Amazon S3 URI for the input data. The S3 bucket must be in the same Region as the API endpoint that you are calling. The URI can point to a single input file or it can provide the prefix for a collection of input files.
For example, if you use the URI S3://bucketName/prefix, if the prefix is a single file, Amazon Comprehend uses that file as input. If more than one file begins with the prefix, Amazon Comprehend uses all of them as input.
This parameter is required if you set DataFormat to COMPREHEND_CSV.
Describes the dataset input data configuration for a document classifier model.
For more information on how the input file is formatted, see Preparing training data in the Comprehend Developer Guide.
AttributeNamesListItem: type: string pattern: ^[a-zA-Z0-9](-*[a-zA-Z0-9])* minLength: 1 maxLength: 63 DatasetDataFormat: type: string enum: - COMPREHEND_CSV - AUGMENTED_MANIFEST AttributeNamesList: type: array items: $ref: '#/components/schemas/AttributeNamesListItem' DatasetType: type: string enum: - TRAIN - TEST DatasetEntityRecognizerEntityList: type: object required: - S3Uri properties: S3Uri: allOf: - $ref: '#/components/schemas/S3Uri' - description: Specifies the Amazon S3 location where the entity list is located. description:Describes the dataset entity list for an entity recognizer model.
For more information on how the input file is formatted, see Preparing training data in the Comprehend Developer Guide.
InvalidRequestException: {} DatasetAugmentedManifestsList: type: array items: $ref: '#/components/schemas/DatasetAugmentedManifestsListItem' LabelDelimiter: type: string pattern: ^[ ~!@#$%^*\-_+=|\\:;\t>?/]$ minLength: 1 maxLength: 1 ComprehendFlywheelArn: type: string pattern: arn:aws(-[^:]+)?:comprehend:[a-zA-Z0-9-]*:[0-9]{12}:flywheel/[a-zA-Z0-9](-*[a-zA-Z0-9])* maxLength: 256 ResourceInUseException: {} TagKey: type: string minLength: 1 maxLength: 128 DatasetEntityRecognizerInputDataConfig: type: object required: - Documents properties: Annotations: allOf: - $ref: '#/components/schemas/DatasetEntityRecognizerAnnotations' - description: The S3 location of the annotation documents for your custom entity recognizer. Documents: allOf: - $ref: '#/components/schemas/DatasetEntityRecognizerDocuments' - description: The format and location of the training documents for your custom entity recognizer. EntityList: allOf: - $ref: '#/components/schemas/DatasetEntityRecognizerEntityList' - description: The S3 location of the entity list for your custom entity recognizer. description: Specifies the format and location of the input data. You must provide either theAnnotations parameter or the EntityList parameter.
TooManyRequestsException: {}
TagValue:
type: string
minLength: 0
maxLength: 256
InternalServerException: {}
ComprehendDatasetArn:
type: string
pattern: arn:aws(-[^:]+)?:comprehend:[a-zA-Z0-9-]*:[0-9]{12}:flywheel/[a-zA-Z0-9](-*[a-zA-Z0-9])*/dataset/[a-zA-Z0-9](-*[a-zA-Z0-9])*
maxLength: 256
ComprehendArnName:
type: string
pattern: ^[a-zA-Z0-9](-*[a-zA-Z0-9])*$
maxLength: 63
Description:
type: string
pattern: ^([a-zA-Z0-9_])[\\a-zA-Z0-9_@#%*+=:?./!\s-]*$
maxLength: 2048
AugmentedManifestsDocumentTypeFormat:
type: string
enum:
- PLAIN_TEXT_DOCUMENT
- SEMI_STRUCTURED_DOCUMENT
Tag:
type: object
required:
- Key
properties:
Key:
allOf:
- $ref: '#/components/schemas/TagKey'
- description: 'The initial part of a key-value pair that forms a tag associated with a given resource. For instance, if you want to show which resources are used by which departments, you might use “Department” as the key portion of the pair, with multiple possible values such as “sales,” “legal,” and “administration.” '
Value:
allOf:
- $ref: '#/components/schemas/TagValue'
- description: ' The second part of a key-value pair that forms a tag associated with a given resource. For instance, if you want to show which resources are used by which departments, you might use “Department” as the initial (key) portion of the pair, with a value of “sales” to indicate the sales department. '
description: 'A key-value pair that adds as a metadata to a resource used by Amazon Comprehend. For example, a tag with the key-value pair ‘Department’:’Sales’ might be added to a resource to indicate its use by a particular department. '
TagList:
type: array
items:
$ref: '#/components/schemas/Tag'
parameters:
X-Amz-Date:
name: X-Amz-Date
in: header
schema:
type: string
required: false
X-Amz-SignedHeaders:
name: X-Amz-SignedHeaders
in: header
schema:
type: string
required: false
X-Amz-Credential:
name: X-Amz-Credential
in: header
schema:
type: string
required: false
X-Amz-Content-Sha256:
name: X-Amz-Content-Sha256
in: header
schema:
type: string
required: false
X-Amz-Algorithm:
name: X-Amz-Algorithm
in: header
schema:
type: string
required: false
X-Amz-Signature:
name: X-Amz-Signature
in: header
schema:
type: string
required: false
X-Amz-Security-Token:
name: X-Amz-Security-Token
in: header
schema:
type: string
required: false
securitySchemes:
hmac:
type: apiKey
name: Authorization
in: header
description: Amazon Signature authorization v4
x-amazon-apigateway-authtype: awsSigv4
externalDocs:
description: Amazon Web Services documentation
url: https://docs.aws.amazon.com/comprehend/
x-hasEquivalentPaths: true