openapi: 3.0.0
info:
version: '2017-11-27'
x-release: v4
title: 'Amazon Comprehend #X Amz Target=Comprehend 20171127.BatchDetectDominantLanguage #X Amz Target=Comprehend 20171127.BatchDetectDominantLanguage #X Amz Target=Comprehend 20171127.StartDocumentClassificationJob API'
description: Amazon Comprehend is an Amazon Web Services service for gaining insight into the content of documents. Use these actions to determine the topics contained in your documents, the topics they discuss, the predominant sentiment expressed in them, the predominant language used, and more.
x-logo:
url: https://twitter.com/awscloud/profile_image?size=original
backgroundColor: '#FFFFFF'
termsOfService: https://aws.amazon.com/service-terms/
contact:
name: Mike Ralphson
email: mike.ralphson@gmail.com
url: https://github.com/mermade/aws2openapi
x-twitter: PermittedSoc
license:
name: Apache 2.0 License
url: http://www.apache.org/licenses/
x-providerName: amazonaws.com
x-serviceName: comprehend
x-aws-signingName: comprehend
x-origin:
- contentType: application/json
url: https://raw.githubusercontent.com/aws/aws-sdk-js/master/apis/comprehend-2017-11-27.normal.json
converter:
url: https://github.com/mermade/aws2openapi
version: 1.0.0
x-apisguru-driver: external
x-apiClientRegistration:
url: https://portal.aws.amazon.com/gp/aws/developer/registration/index.html?nc2=h_ct
x-apisguru-categories:
- cloud
x-preferred: true
servers:
- url: http://comprehend.{region}.amazonaws.com
variables:
region:
description: The AWS region
enum:
- us-east-1
- us-east-2
- us-west-1
- us-west-2
- us-gov-west-1
- us-gov-east-1
- ca-central-1
- eu-north-1
- eu-west-1
- eu-west-2
- eu-west-3
- eu-central-1
- eu-south-1
- af-south-1
- ap-northeast-1
- ap-northeast-2
- ap-northeast-3
- ap-southeast-1
- ap-southeast-2
- ap-east-1
- ap-south-1
- sa-east-1
- me-south-1
default: us-east-1
description: The Amazon Comprehend multi-region endpoint
- url: https://comprehend.{region}.amazonaws.com
variables:
region:
description: The AWS region
enum:
- us-east-1
- us-east-2
- us-west-1
- us-west-2
- us-gov-west-1
- us-gov-east-1
- ca-central-1
- eu-north-1
- eu-west-1
- eu-west-2
- eu-west-3
- eu-central-1
- eu-south-1
- af-south-1
- ap-northeast-1
- ap-northeast-2
- ap-northeast-3
- ap-southeast-1
- ap-southeast-2
- ap-east-1
- ap-south-1
- sa-east-1
- me-south-1
default: us-east-1
description: The Amazon Comprehend multi-region endpoint
- url: http://comprehend.{region}.amazonaws.com.cn
variables:
region:
description: The AWS region
enum:
- cn-north-1
- cn-northwest-1
default: cn-north-1
description: The Amazon Comprehend endpoint for China (Beijing) and China (Ningxia)
- url: https://comprehend.{region}.amazonaws.com.cn
variables:
region:
description: The AWS region
enum:
- cn-north-1
- cn-northwest-1
default: cn-north-1
description: The Amazon Comprehend endpoint for China (Beijing) and China (Ningxia)
security:
- hmac: []
tags:
- name: '#X Amz Target=Comprehend 20171127.StartDocumentClassificationJob'
paths:
/#X-Amz-Target=Comprehend_20171127.StartDocumentClassificationJob:
parameters:
- $ref: '#/components/parameters/X-Amz-Content-Sha256'
- $ref: '#/components/parameters/X-Amz-Date'
- $ref: '#/components/parameters/X-Amz-Algorithm'
- $ref: '#/components/parameters/X-Amz-Credential'
- $ref: '#/components/parameters/X-Amz-Security-Token'
- $ref: '#/components/parameters/X-Amz-Signature'
- $ref: '#/components/parameters/X-Amz-SignedHeaders'
post:
operationId: StartDocumentClassificationJob
description: Starts an asynchronous document classification job. Use the DescribeDocumentClassificationJob operation to track the progress of the job.
responses:
'200':
description: Success
content:
application/json:
schema:
$ref: '#/components/schemas/StartDocumentClassificationJobResponse'
examples:
StartDocumentClassificationJob200Example:
summary: Default StartDocumentClassificationJob 200
x-microcks-default: true
value:
JobId: example
JobArn: example
JobStatus: example
DocumentClassifierArn: example
'480':
description: InvalidRequestException
content:
application/json:
schema:
$ref: '#/components/schemas/InvalidRequestException'
examples:
StartDocumentClassificationJob480Example:
summary: Default StartDocumentClassificationJob 480
x-microcks-default: true
value: example
'481':
description: TooManyRequestsException
content:
application/json:
schema:
$ref: '#/components/schemas/TooManyRequestsException'
examples:
StartDocumentClassificationJob481Example:
summary: Default StartDocumentClassificationJob 481
x-microcks-default: true
value: example
'482':
description: ResourceNotFoundException
content:
application/json:
schema:
$ref: '#/components/schemas/ResourceNotFoundException'
examples:
StartDocumentClassificationJob482Example:
summary: Default StartDocumentClassificationJob 482
x-microcks-default: true
value: example
'483':
description: ResourceUnavailableException
content:
application/json:
schema:
$ref: '#/components/schemas/ResourceUnavailableException'
examples:
StartDocumentClassificationJob483Example:
summary: Default StartDocumentClassificationJob 483
x-microcks-default: true
value: example
'484':
description: KmsKeyValidationException
content:
application/json:
schema:
$ref: '#/components/schemas/KmsKeyValidationException'
examples:
StartDocumentClassificationJob484Example:
summary: Default StartDocumentClassificationJob 484
x-microcks-default: true
value: example
'485':
description: TooManyTagsException
content:
application/json:
schema:
$ref: '#/components/schemas/TooManyTagsException'
examples:
StartDocumentClassificationJob485Example:
summary: Default StartDocumentClassificationJob 485
x-microcks-default: true
value: example
'486':
description: ResourceInUseException
content:
application/json:
schema:
$ref: '#/components/schemas/ResourceInUseException'
examples:
StartDocumentClassificationJob486Example:
summary: Default StartDocumentClassificationJob 486
x-microcks-default: true
value: example
'487':
description: InternalServerException
content:
application/json:
schema:
$ref: '#/components/schemas/InternalServerException'
examples:
StartDocumentClassificationJob487Example:
summary: Default StartDocumentClassificationJob 487
x-microcks-default: true
value: example
requestBody:
required: true
content:
application/json:
schema:
$ref: '#/components/schemas/StartDocumentClassificationJobRequest'
parameters:
- name: X-Amz-Target
in: header
required: true
schema:
type: string
enum:
- Comprehend_20171127.StartDocumentClassificationJob
summary: Amazon Comprehend Start Document Classification Job
x-microcks-operation:
delay: 0
dispatcher: FALLBACK
tags:
- '#X Amz Target=Comprehend 20171127.StartDocumentClassificationJob'
components:
schemas:
IamRoleArn:
type: string
pattern: arn:aws(-[^:]+)?:iam::[0-9]{12}:role/.+
minLength: 20
maxLength: 2048
SecurityGroupId:
type: string
pattern: '[-0-9a-zA-Z]+'
minLength: 1
maxLength: 32
SubnetId:
type: string
pattern: '[-0-9a-zA-Z]+'
minLength: 1
maxLength: 32
ClientRequestTokenString:
type: string
pattern: ^[a-zA-Z0-9-]+$
minLength: 1
maxLength: 64
DocumentReadMode:
type: string
enum:
- SERVICE_DEFAULT
- FORCE_DOCUMENT_READ_ACTION
TooManyTagsException: {}
ResourceNotFoundException: {}
JobStatus:
type: string
enum:
- SUBMITTED
- IN_PROGRESS
- COMPLETED
- FAILED
- STOP_REQUESTED
- STOPPED
KmsKeyId:
type: string
pattern: ^\p{ASCII}+$
maxLength: 2048
S3Uri:
type: string
pattern: s3://[a-z0-9][\.\-a-z0-9]{1,61}[a-z0-9](/.*)?
maxLength: 1024
Subnets:
type: array
items:
$ref: '#/components/schemas/SubnetId'
minItems: 1
maxItems: 16
DocumentReaderConfig:
type: object
required:
- DocumentReadAction
properties:
DocumentReadAction:
allOf:
- $ref: '#/components/schemas/DocumentReadAction'
- description:
This field defines the Amazon Textract API operation that Amazon Comprehend uses to extract text from PDF files and image files. Enter one of the following values:
TEXTRACT_DETECT_DOCUMENT_TEXT - The Amazon Comprehend service uses the DetectDocumentText API operation.
TEXTRACT_ANALYZE_DOCUMENT - The Amazon Comprehend service uses the AnalyzeDocument API operation.
Determines the text extraction actions for PDF files. Enter one of the following values:
SERVICE_DEFAULT - use the Amazon Comprehend service defaults for PDF files.
FORCE_DOCUMENT_READ_ACTION - Amazon Comprehend uses the Textract API specified by DocumentReadAction for all PDF files, including digital PDF files.
Specifies the type of Amazon Textract features to apply. If you chose TEXTRACT_ANALYZE_DOCUMENT as the read action, you must specify one or both of the following values:
TABLES - Returns information about any tables that are detected in the input document.
FORMS - Returns information and the data from any forms that are detected in the input document.
Provides configuration parameters to override the default actions for extracting text from PDF documents and image files.
By default, Amazon Comprehend performs the following actions to extract text from files, based on the input file type:
Word files - Amazon Comprehend parser extracts the text.
Digital PDF files - Amazon Comprehend parser extracts the text.
Image files and scanned PDF files - Amazon Comprehend uses the Amazon Textract DetectDocumentText API to extract the text.
DocumentReaderConfig does not apply to plain text files or Word files.
For image files and PDF documents, you can override these default actions using the fields listed below. For more information, see Setting text extraction options in the Comprehend Developer Guide.
' InputFormat: type: string enum: - ONE_DOC_PER_FILE - ONE_DOC_PER_LINE JobId: type: string pattern: ^([\p{L}\p{Z}\p{N}_.:/=+\-%@]*)$ minLength: 1 maxLength: 32 StartDocumentClassificationJobRequest: type: object required: - InputDataConfig - OutputDataConfig - DataAccessRoleArn title: StartDocumentClassificationJobRequest properties: JobName: allOf: - $ref: '#/components/schemas/JobName' - description: The identifier of the job. DocumentClassifierArn: allOf: - $ref: '#/components/schemas/DocumentClassifierArn' - description: The Amazon Resource Name (ARN) of the document classifier to use to process the job. InputDataConfig: allOf: - $ref: '#/components/schemas/InputDataConfig' - description: Specifies the format and location of the input data for the job. OutputDataConfig: allOf: - $ref: '#/components/schemas/OutputDataConfig' - description: Specifies where to send the output files. DataAccessRoleArn: allOf: - $ref: '#/components/schemas/IamRoleArn' - description: The Amazon Resource Name (ARN) of the IAM role that grants Amazon Comprehend read access to your input data. ClientRequestToken: allOf: - $ref: '#/components/schemas/ClientRequestTokenString' - description: A unique identifier for the request. If you do not set the client request token, Amazon Comprehend generates one. VolumeKmsKeyId: allOf: - $ref: '#/components/schemas/KmsKeyId' - description: 'ID for the Amazon Web Services Key Management Service (KMS) key that Amazon Comprehend uses to encrypt data on the storage volume attached to the ML compute instance(s) that process the analysis job. The VolumeKmsKeyId can be either of the following formats:
KMS Key ID: "1234abcd-12ab-34cd-56ef-1234567890ab"
Amazon Resource Name (ARN) of a KMS Key: "arn:aws:kms:us-west-2:111122223333:key/1234abcd-12ab-34cd-56ef-1234567890ab"
When you use the OutputDataConfig object with asynchronous operations, you specify the Amazon S3 location where you want to write the output data. The URI must be in the same Region as the API endpoint that you are calling. The location is used as the prefix for the actual location of the output file.
When the topic detection job is finished, the service creates an output file in a directory specific to the job. The S3Uri field contains the location of the output file, called output.tar.gz. It is a compressed archive that contains the ouput of the operation.
For a PII entity detection job, the output file is plain text, not a compressed archive. The output file name is the same as the input file, with .out appended at the end.
ID for the Amazon Web Services Key Management Service (KMS) key that Amazon Comprehend uses to encrypt the output results from an analysis job. The KmsKeyId can be one of the following formats:
KMS Key ID: "1234abcd-12ab-34cd-56ef-1234567890ab"
Amazon Resource Name (ARN) of a KMS Key: "arn:aws:kms:us-west-2:111122223333:key/1234abcd-12ab-34cd-56ef-1234567890ab"
KMS Key Alias: "alias/ExampleAlias"
ARN of a KMS Key Alias: "arn:aws:kms:us-west-2:111122223333:alias/ExampleAlias"
Provides configuration parameters for the output of inference jobs.
DocumentClassifierArn: type: string pattern: arn:aws(-[^:]+)?:comprehend:[a-zA-Z0-9-]*:[0-9]{12}:document-classifier/[a-zA-Z0-9](-*[a-zA-Z0-9])*(/version/[a-zA-Z0-9](-*[a-zA-Z0-9])*)? maxLength: 256 ResourceUnavailableException: {} JobName: type: string pattern: ^([\p{L}\p{Z}\p{N}_.:/=+\-%@]*)$ minLength: 1 maxLength: 256 InvalidRequestException: {} DocumentReadFeatureTypes: type: string enum: - TABLES - FORMS description:Specifies the type of Amazon Textract features to apply. If you chose TEXTRACT_ANALYZE_DOCUMENT as the read action, you must specify one or both of the following values:
TABLES - Returns additional information about any tables that are detected in the input document.
FORMS - Returns additional information about any forms that are detected in the input document.
The Amazon S3 URI for the input data. The URI must be in same Region as the API endpoint that you are calling. The URI can point to a single input file or it can provide the prefix for a collection of data files.
For example, if you use the URI S3://bucketName/prefix, if the prefix is a single file, Amazon Comprehend uses that file as input. If more than one file begins with the prefix, Amazon Comprehend uses all of them as input.
Specifies how the text in an input file should be processed:
ONE_DOC_PER_FILE - Each file is considered a separate document. Use this option when you are processing large documents, such as newspaper articles or scientific papers.
ONE_DOC_PER_LINE - Each line in a file is considered a separate document. Use this option when you are processing many short documents, such as text messages.
DescribeDocumentClassificationJob operation.
JobArn:
allOf:
- $ref: '#/components/schemas/ComprehendArn'
- description: The Amazon Resource Name (ARN) of the document classification job. It is a unique, fully qualified identifier for the job. It includes the Amazon Web Services account, Amazon Web Services Region, and the job ID. The format of the ARN is as follows:
arn:<partition>:comprehend:<region>:<account-id>:document-classification-job/<job-id>
The following is an example job ARN:
arn:aws:comprehend:us-west-2:111122223333:document-classification-job/1234abcd12ab34cd56ef1234567890ab
The status of the job:
SUBMITTED - The job has been received and queued for processing.
IN_PROGRESS - Amazon Comprehend is processing the job.
COMPLETED - The job was successfully completed and the output is available.
FAILED - The job did not complete. For details, use the DescribeDocumentClassificationJob operation.
STOP_REQUESTED - Amazon Comprehend has received a stop request for the job and is processing the request.
STOPPED - The job was successfully stopped without completing.