import * as pulumi from "@pulumi/pulumi"; import * as inputs from "../types/input"; import * as outputs from "../types/output"; /** * Manages an AWS SageMaker AI Training Job. * * ## Example Usage * * ### Basic Usage * * ```typescript * import * as pulumi from "@pulumi/pulumi"; * import * as aws from "@pulumi/aws"; * * const example = new aws.sagemaker.TrainingJob("example", { * algorithmSpecification: { * trainingInputMode: "File", * trainingImage: exampleAwsSagemakerPrebuiltEcrImage.registryPath, * }, * outputDataConfig: { * s3OutputPath: `s3://${exampleAwsS3Bucket.bucket}/output/`, * }, * resourceConfig: { * instanceType: "ml.m5.large", * instanceCount: 1, * volumeSizeInGb: 30, * }, * stoppingCondition: { * maxRuntimeInSeconds: 3600, * }, * trainingJobName: "example", * roleArn: exampleAwsIamRole.arn, * }); * ``` * * ### With VPC Configuration * * ```typescript * import * as pulumi from "@pulumi/pulumi"; * import * as aws from "@pulumi/aws"; * * const example = new aws.sagemaker.TrainingJob("example", { * algorithmSpecification: { * trainingInputMode: "File", * trainingImage: exampleAwsSagemakerPrebuiltEcrImage.registryPath, * }, * outputDataConfig: { * s3OutputPath: `s3://${exampleAwsS3Bucket.bucket}/output/`, * }, * resourceConfig: { * instanceType: "ml.m5.large", * instanceCount: 1, * volumeSizeInGb: 30, * }, * stoppingCondition: { * maxRuntimeInSeconds: 3600, * }, * vpcConfig: { * securityGroupIds: [exampleAwsSecurityGroup.id], * subnets: [exampleAwsSubnet.id], * }, * trainingJobName: "example", * roleArn: exampleAwsIamRole.arn, * }); * ``` * * ### With Input Data and Hyperparameters * * ```typescript * import * as pulumi from "@pulumi/pulumi"; * import * as aws from "@pulumi/aws"; * * const example = new aws.sagemaker.TrainingJob("example", { * algorithmSpecification: { * trainingInputMode: "File", * trainingImage: exampleAwsSagemakerPrebuiltEcrImage.registryPath, * enableSagemakerMetricsTimeSeries: true, * }, * outputDataConfig: { * s3OutputPath: `s3://${exampleAwsS3Bucket.bucket}/output/`, * }, * resourceConfig: { * instanceType: "ml.m5.large", * instanceCount: 1, * volumeSizeInGb: 30, * }, * stoppingCondition: { * maxRuntimeInSeconds: 3600, * }, * inputDataConfigs: [{ * dataSource: { * s3DataSource: { * s3DataType: "S3Prefix", * s3Uri: `s3://${exampleAwsS3Bucket.bucket}/train/`, * }, * }, * channelName: "train", * }], * trainingJobName: "example", * roleArn: exampleAwsIamRole.arn, * hyperParameters: { * mini_batch_size: "200", * epochs: "10", * }, * }); * ``` * * ### With Encrypted Output, Checkpoints, and TensorBoard * * ```typescript * import * as pulumi from "@pulumi/pulumi"; * import * as aws from "@pulumi/aws"; * * const example = new aws.sagemaker.TrainingJob("example", { * algorithmSpecification: { * trainingInputMode: "File", * trainingImage: exampleAwsSagemakerPrebuiltEcrImage.registryPath, * }, * checkpointConfig: { * localPath: "/opt/ml/checkpoints", * s3Uri: `s3://${exampleAwsS3Bucket.bucket}/checkpoints/`, * }, * outputDataConfig: { * compressionType: "GZIP", * kmsKeyId: exampleAwsKmsKey.arn, * s3OutputPath: `s3://${exampleAwsS3Bucket.bucket}/output/`, * }, * resourceConfig: { * instanceType: "ml.m5.large", * instanceCount: 1, * volumeSizeInGb: 30, * volumeKmsKeyId: exampleAwsKmsKey.arn, * }, * stoppingCondition: { * maxRuntimeInSeconds: 3600, * }, * tensorBoardOutputConfig: { * localPath: "/opt/ml/output/tensorboard", * s3OutputPath: `s3://${exampleAwsS3Bucket.bucket}/tensorboard/`, * }, * trainingJobName: "example", * roleArn: exampleAwsIamRole.arn, * }); * ``` * * ### With Managed Spot Training and Custom Metrics * * ```typescript * import * as pulumi from "@pulumi/pulumi"; * import * as aws from "@pulumi/aws"; * * const example = new aws.sagemaker.TrainingJob("example", { * algorithmSpecification: { * metricDefinitions: [ * { * name: "train:loss", * regex: "loss: ([0-9\\.]+)", * }, * { * name: "validation:accuracy", * regex: "accuracy: ([0-9\\.]+)", * }, * ], * trainingInputMode: "File", * trainingImage: trainingImage, * containerEntrypoints: [ * "python", * "/opt/ml/code/train.py", * ], * containerArguments: [ * "--epochs", * "10", * "--batch-size", * "128", * ], * }, * outputDataConfig: { * s3OutputPath: `s3://${exampleAwsS3Bucket.bucket}/output/`, * }, * resourceConfig: { * instanceType: "ml.m5.xlarge", * instanceCount: 1, * volumeSizeInGb: 50, * keepAlivePeriodInSeconds: 600, * }, * retryStrategy: { * maximumRetryAttempts: 3, * }, * stoppingCondition: { * maxRuntimeInSeconds: 3600, * maxWaitTimeInSeconds: 7200, * }, * trainingJobName: "example", * roleArn: exampleAwsIamRole.arn, * enableManagedSpotTraining: true, * enableNetworkIsolation: true, * enableInterContainerTrafficEncryption: true, * environment: { * MODEL_DIR: "/opt/ml/model", * SM_LOG_LEVEL: "20", * }, * hyperParameters: { * epochs: "10", * batch_size: "128", * }, * tags: { * Environment: "test", * Workload: "training", * }, * }); * ``` * * ### With Multiple Input Channels, Infrastructure Checks, and Session Tag Chaining * * ```typescript * import * as pulumi from "@pulumi/pulumi"; * import * as aws from "@pulumi/aws"; * * const example = new aws.sagemaker.TrainingJob("example", { * algorithmSpecification: { * trainingInputMode: "File", * trainingImage: exampleAwsSagemakerPrebuiltEcrImage.registryPath, * }, * infraCheckConfig: { * enableInfraCheck: true, * }, * outputDataConfig: { * s3OutputPath: `s3://${exampleAwsS3Bucket.bucket}/output/`, * }, * resourceConfig: { * instanceType: "ml.m5.large", * instanceCount: 1, * volumeSizeInGb: 30, * }, * sessionChainingConfig: { * enableSessionTagChaining: true, * }, * stoppingCondition: { * maxRuntimeInSeconds: 3600, * }, * inputDataConfigs: [ * { * dataSource: { * s3DataSource: { * s3DataDistributionType: "FullyReplicated", * s3DataType: "S3Prefix", * s3Uri: `s3://${exampleAwsS3Bucket.bucket}/train/`, * }, * }, * channelName: "train", * contentType: "text/csv", * inputMode: "File", * }, * { * dataSource: { * s3DataSource: { * s3DataDistributionType: "FullyReplicated", * s3DataType: "S3Prefix", * s3Uri: `s3://${exampleAwsS3Bucket.bucket}/validation/`, * }, * }, * channelName: "validation", * contentType: "text/csv", * inputMode: "File", * }, * ], * trainingJobName: "example", * roleArn: exampleAwsIamRole.arn, * }); * ``` * * ## Import * * ### Identity Schema * * #### Required * * * `trainingJobName` - (String) Name of the Training Job. * * #### Optional * * * `accountId` (String) AWS Account where this resource is managed. * * `region` (String) Region where this resource is managed. * * Using `pulumi import`, import SageMaker AI Training Job using the `trainingJobName`. For example: * * ```sh * $ pulumi import aws:sagemaker/trainingJob:TrainingJob example my-training-job * ``` */ export declare class TrainingJob extends pulumi.CustomResource { /** * Get an existing TrainingJob resource's state with the given name, ID, and optional extra * properties used to qualify the lookup. * * @param name The _unique_ name of the resulting resource. * @param id The _unique_ provider ID of the resource to lookup. * @param state Any extra arguments used during the lookup. * @param opts Optional settings to control the behavior of the CustomResource. */ static get(name: string, id: pulumi.Input, state?: TrainingJobState, opts?: pulumi.CustomResourceOptions): TrainingJob; /** * Returns true if the given object is an instance of TrainingJob. This is designed to work even * when multiple copies of the Pulumi SDK have been loaded into the same process. */ static isInstance(obj: any): obj is TrainingJob; /** * Algorithm-related parameters of the training job. See `algorithmSpecification` below. Conflicts with `serverlessJobConfig`. */ readonly algorithmSpecification: pulumi.Output; /** * ARN of the Training Job. */ readonly arn: pulumi.Output; /** * Location of checkpoints during training. See `checkpointConfig` below. Conflicts with `serverlessJobConfig`. */ readonly checkpointConfig: pulumi.Output; /** * Configuration for debugging rules. See `debugHookConfig` below. Conflicts with `serverlessJobConfig`. */ readonly debugHookConfig: pulumi.Output; /** * List of debug rule configurations. Maximum of 20. See `debugRuleConfigurations` below. */ readonly debugRuleConfigurations: pulumi.Output; /** * Whether to delete model packages in the configured model package group when the training job is destroyed. Default is `false`. */ readonly deleteModelPackagesOnDestroy: pulumi.Output; /** * Whether to delete detached VPC ENIs SageMaker may leave behind when the training job is destroyed. Default is `false`. */ readonly deleteVpcEnisOnDestroy: pulumi.Output; /** * Whether to encrypt inter-container traffic. When enabled, communications between containers are encrypted. */ readonly enableInterContainerTrafficEncryption: pulumi.Output; /** * Whether to use managed spot training. Optimizes the cost of training by using Amazon EC2 Spot Instances. Conflicts with `serverlessJobConfig`. */ readonly enableManagedSpotTraining: pulumi.Output; /** * Whether to isolate the training container from the network. No inbound or outbound network calls can be made. */ readonly enableNetworkIsolation: pulumi.Output; /** * Map of environment variables to set in the training container. Maximum of 100 entries. Conflicts with `serverlessJobConfig`. */ readonly environment: pulumi.Output<{ [key: string]: string; } | undefined>; /** * Associates a SageMaker AI Experiment or Trial to the training job. See `experimentConfig` below. Conflicts with `serverlessJobConfig`. */ readonly experimentConfig: pulumi.Output; /** * Map of hyperparameters for the training algorithm. Maximum of 100 entries. */ readonly hyperParameters: pulumi.Output<{ [key: string]: string; } | undefined>; /** * Infrastructure health check configuration. See `infraCheckConfig` below. */ readonly infraCheckConfig: pulumi.Output; /** * List of input data channel configurations for the training job. Maximum of 20. See `inputDataConfig` below. */ readonly inputDataConfigs: pulumi.Output; /** * MLflow integration configuration. See `mlflowConfig` below. */ readonly mlflowConfig: pulumi.Output; /** * Model package configuration. Requires `serverlessJobConfig`. See `modelPackageConfig` below. */ readonly modelPackageConfig: pulumi.Output; /** * Location of the output data from the training job. See `outputDataConfig` below. * * The following arguments are optional: */ readonly outputDataConfig: pulumi.Output; /** * Configuration for the profiler. See `profilerConfig` below. Conflicts with `serverlessJobConfig`. */ readonly profilerConfig: pulumi.Output; /** * List of profiler rule configurations. Maximum of 20. See `profilerRuleConfigurations` below. Conflicts with `serverlessJobConfig`. */ readonly profilerRuleConfigurations: pulumi.Output; /** * Region where this resource will be [managed](https://docs.aws.amazon.com/general/latest/gr/rande.html#regional-endpoints). Defaults to the Region set in the provider configuration. */ readonly region: pulumi.Output; /** * Configuration for remote debugging. See `remoteDebugConfig` below. */ readonly remoteDebugConfig: pulumi.Output; /** * Resources for the training job, including compute instances and storage volumes. See `resourceConfig` below. */ readonly resourceConfig: pulumi.Output; /** * Number of times to retry the job if it fails. See `retryStrategy` below. Conflicts with `serverlessJobConfig`. */ readonly retryStrategy: pulumi.Output; /** * ARN of the IAM role that SageMaker AI assumes to perform tasks on your behalf during training. */ readonly roleArn: pulumi.Output; /** * Configuration for serverless training jobs using foundation models. Conflicts with `algorithmSpecification`, `enableManagedSpotTraining`, `environment`, `retryStrategy`, `checkpointConfig`, `debugHookConfig`, `experimentConfig`, `profilerConfig`, `profilerRuleConfigurations`, and `tensorBoardOutputConfig`. See `serverlessJobConfig` below. */ readonly serverlessJobConfig: pulumi.Output; /** * Configuration for session tag chaining. See `sessionChainingConfig` below. */ readonly sessionChainingConfig: pulumi.Output; readonly stoppingCondition: pulumi.Output; /** * Map of tags to assign to the resource. If configured with a provider `defaultTags` configuration block present, tags with matching keys will overwrite those defined at the provider-level. */ readonly tags: pulumi.Output<{ [key: string]: string; } | undefined>; /** * Map of tags assigned to the resource, including those inherited from the provider `defaultTags` configuration block. */ readonly tagsAll: pulumi.Output<{ [key: string]: string; }>; /** * Configuration for TensorBoard output. See `tensorBoardOutputConfig` below. Conflicts with `serverlessJobConfig`. */ readonly tensorBoardOutputConfig: pulumi.Output; readonly timeouts: pulumi.Output; /** * Name of the training job. Must be between 1 and 63 characters, start with a letter or number, and contain only letters, numbers, and hyphens. */ readonly trainingJobName: pulumi.Output; /** * VPC configuration for the training job. See `vpcConfig` below. */ readonly vpcConfig: pulumi.Output; /** * Create a TrainingJob resource with the given unique name, arguments, and options. * * @param name The _unique_ name of the resource. * @param args The arguments to use to populate this resource's properties. * @param opts A bag of options that control this resource's behavior. */ constructor(name: string, args: TrainingJobArgs, opts?: pulumi.CustomResourceOptions); } /** * Input properties used for looking up and filtering TrainingJob resources. */ export interface TrainingJobState { /** * Algorithm-related parameters of the training job. See `algorithmSpecification` below. Conflicts with `serverlessJobConfig`. */ algorithmSpecification?: pulumi.Input; /** * ARN of the Training Job. */ arn?: pulumi.Input; /** * Location of checkpoints during training. See `checkpointConfig` below. Conflicts with `serverlessJobConfig`. */ checkpointConfig?: pulumi.Input; /** * Configuration for debugging rules. See `debugHookConfig` below. Conflicts with `serverlessJobConfig`. */ debugHookConfig?: pulumi.Input; /** * List of debug rule configurations. Maximum of 20. See `debugRuleConfigurations` below. */ debugRuleConfigurations?: pulumi.Input[] | undefined>; /** * Whether to delete model packages in the configured model package group when the training job is destroyed. Default is `false`. */ deleteModelPackagesOnDestroy?: pulumi.Input; /** * Whether to delete detached VPC ENIs SageMaker may leave behind when the training job is destroyed. Default is `false`. */ deleteVpcEnisOnDestroy?: pulumi.Input; /** * Whether to encrypt inter-container traffic. When enabled, communications between containers are encrypted. */ enableInterContainerTrafficEncryption?: pulumi.Input; /** * Whether to use managed spot training. Optimizes the cost of training by using Amazon EC2 Spot Instances. Conflicts with `serverlessJobConfig`. */ enableManagedSpotTraining?: pulumi.Input; /** * Whether to isolate the training container from the network. No inbound or outbound network calls can be made. */ enableNetworkIsolation?: pulumi.Input; /** * Map of environment variables to set in the training container. Maximum of 100 entries. Conflicts with `serverlessJobConfig`. */ environment?: pulumi.Input<{ [key: string]: pulumi.Input; } | undefined>; /** * Associates a SageMaker AI Experiment or Trial to the training job. See `experimentConfig` below. Conflicts with `serverlessJobConfig`. */ experimentConfig?: pulumi.Input; /** * Map of hyperparameters for the training algorithm. Maximum of 100 entries. */ hyperParameters?: pulumi.Input<{ [key: string]: pulumi.Input; } | undefined>; /** * Infrastructure health check configuration. See `infraCheckConfig` below. */ infraCheckConfig?: pulumi.Input; /** * List of input data channel configurations for the training job. Maximum of 20. See `inputDataConfig` below. */ inputDataConfigs?: pulumi.Input[] | undefined>; /** * MLflow integration configuration. See `mlflowConfig` below. */ mlflowConfig?: pulumi.Input; /** * Model package configuration. Requires `serverlessJobConfig`. See `modelPackageConfig` below. */ modelPackageConfig?: pulumi.Input; /** * Location of the output data from the training job. See `outputDataConfig` below. * * The following arguments are optional: */ outputDataConfig?: pulumi.Input; /** * Configuration for the profiler. See `profilerConfig` below. Conflicts with `serverlessJobConfig`. */ profilerConfig?: pulumi.Input; /** * List of profiler rule configurations. Maximum of 20. See `profilerRuleConfigurations` below. Conflicts with `serverlessJobConfig`. */ profilerRuleConfigurations?: pulumi.Input[] | undefined>; /** * Region where this resource will be [managed](https://docs.aws.amazon.com/general/latest/gr/rande.html#regional-endpoints). Defaults to the Region set in the provider configuration. */ region?: pulumi.Input; /** * Configuration for remote debugging. See `remoteDebugConfig` below. */ remoteDebugConfig?: pulumi.Input; /** * Resources for the training job, including compute instances and storage volumes. See `resourceConfig` below. */ resourceConfig?: pulumi.Input; /** * Number of times to retry the job if it fails. See `retryStrategy` below. Conflicts with `serverlessJobConfig`. */ retryStrategy?: pulumi.Input; /** * ARN of the IAM role that SageMaker AI assumes to perform tasks on your behalf during training. */ roleArn?: pulumi.Input; /** * Configuration for serverless training jobs using foundation models. Conflicts with `algorithmSpecification`, `enableManagedSpotTraining`, `environment`, `retryStrategy`, `checkpointConfig`, `debugHookConfig`, `experimentConfig`, `profilerConfig`, `profilerRuleConfigurations`, and `tensorBoardOutputConfig`. See `serverlessJobConfig` below. */ serverlessJobConfig?: pulumi.Input; /** * Configuration for session tag chaining. See `sessionChainingConfig` below. */ sessionChainingConfig?: pulumi.Input; stoppingCondition?: pulumi.Input; /** * Map of tags to assign to the resource. If configured with a provider `defaultTags` configuration block present, tags with matching keys will overwrite those defined at the provider-level. */ tags?: pulumi.Input<{ [key: string]: pulumi.Input; } | undefined>; /** * Map of tags assigned to the resource, including those inherited from the provider `defaultTags` configuration block. */ tagsAll?: pulumi.Input<{ [key: string]: pulumi.Input; } | undefined>; /** * Configuration for TensorBoard output. See `tensorBoardOutputConfig` below. Conflicts with `serverlessJobConfig`. */ tensorBoardOutputConfig?: pulumi.Input; timeouts?: pulumi.Input; /** * Name of the training job. Must be between 1 and 63 characters, start with a letter or number, and contain only letters, numbers, and hyphens. */ trainingJobName?: pulumi.Input; /** * VPC configuration for the training job. See `vpcConfig` below. */ vpcConfig?: pulumi.Input; } /** * The set of arguments for constructing a TrainingJob resource. */ export interface TrainingJobArgs { /** * Algorithm-related parameters of the training job. See `algorithmSpecification` below. Conflicts with `serverlessJobConfig`. */ algorithmSpecification?: pulumi.Input; /** * Location of checkpoints during training. See `checkpointConfig` below. Conflicts with `serverlessJobConfig`. */ checkpointConfig?: pulumi.Input; /** * Configuration for debugging rules. See `debugHookConfig` below. Conflicts with `serverlessJobConfig`. */ debugHookConfig?: pulumi.Input; /** * List of debug rule configurations. Maximum of 20. See `debugRuleConfigurations` below. */ debugRuleConfigurations?: pulumi.Input[] | undefined>; /** * Whether to delete model packages in the configured model package group when the training job is destroyed. Default is `false`. */ deleteModelPackagesOnDestroy?: pulumi.Input; /** * Whether to delete detached VPC ENIs SageMaker may leave behind when the training job is destroyed. Default is `false`. */ deleteVpcEnisOnDestroy?: pulumi.Input; /** * Whether to encrypt inter-container traffic. When enabled, communications between containers are encrypted. */ enableInterContainerTrafficEncryption?: pulumi.Input; /** * Whether to use managed spot training. Optimizes the cost of training by using Amazon EC2 Spot Instances. Conflicts with `serverlessJobConfig`. */ enableManagedSpotTraining?: pulumi.Input; /** * Whether to isolate the training container from the network. No inbound or outbound network calls can be made. */ enableNetworkIsolation?: pulumi.Input; /** * Map of environment variables to set in the training container. Maximum of 100 entries. Conflicts with `serverlessJobConfig`. */ environment?: pulumi.Input<{ [key: string]: pulumi.Input; } | undefined>; /** * Associates a SageMaker AI Experiment or Trial to the training job. See `experimentConfig` below. Conflicts with `serverlessJobConfig`. */ experimentConfig?: pulumi.Input; /** * Map of hyperparameters for the training algorithm. Maximum of 100 entries. */ hyperParameters?: pulumi.Input<{ [key: string]: pulumi.Input; } | undefined>; /** * Infrastructure health check configuration. See `infraCheckConfig` below. */ infraCheckConfig?: pulumi.Input; /** * List of input data channel configurations for the training job. Maximum of 20. See `inputDataConfig` below. */ inputDataConfigs?: pulumi.Input[] | undefined>; /** * MLflow integration configuration. See `mlflowConfig` below. */ mlflowConfig?: pulumi.Input; /** * Model package configuration. Requires `serverlessJobConfig`. See `modelPackageConfig` below. */ modelPackageConfig?: pulumi.Input; /** * Location of the output data from the training job. See `outputDataConfig` below. * * The following arguments are optional: */ outputDataConfig?: pulumi.Input; /** * Configuration for the profiler. See `profilerConfig` below. Conflicts with `serverlessJobConfig`. */ profilerConfig?: pulumi.Input; /** * List of profiler rule configurations. Maximum of 20. See `profilerRuleConfigurations` below. Conflicts with `serverlessJobConfig`. */ profilerRuleConfigurations?: pulumi.Input[] | undefined>; /** * Region where this resource will be [managed](https://docs.aws.amazon.com/general/latest/gr/rande.html#regional-endpoints). Defaults to the Region set in the provider configuration. */ region?: pulumi.Input; /** * Configuration for remote debugging. See `remoteDebugConfig` below. */ remoteDebugConfig?: pulumi.Input; /** * Resources for the training job, including compute instances and storage volumes. See `resourceConfig` below. */ resourceConfig?: pulumi.Input; /** * Number of times to retry the job if it fails. See `retryStrategy` below. Conflicts with `serverlessJobConfig`. */ retryStrategy?: pulumi.Input; /** * ARN of the IAM role that SageMaker AI assumes to perform tasks on your behalf during training. */ roleArn: pulumi.Input; /** * Configuration for serverless training jobs using foundation models. Conflicts with `algorithmSpecification`, `enableManagedSpotTraining`, `environment`, `retryStrategy`, `checkpointConfig`, `debugHookConfig`, `experimentConfig`, `profilerConfig`, `profilerRuleConfigurations`, and `tensorBoardOutputConfig`. See `serverlessJobConfig` below. */ serverlessJobConfig?: pulumi.Input; /** * Configuration for session tag chaining. See `sessionChainingConfig` below. */ sessionChainingConfig?: pulumi.Input; stoppingCondition?: pulumi.Input; /** * Map of tags to assign to the resource. If configured with a provider `defaultTags` configuration block present, tags with matching keys will overwrite those defined at the provider-level. */ tags?: pulumi.Input<{ [key: string]: pulumi.Input; } | undefined>; /** * Configuration for TensorBoard output. See `tensorBoardOutputConfig` below. Conflicts with `serverlessJobConfig`. */ tensorBoardOutputConfig?: pulumi.Input; timeouts?: pulumi.Input; /** * Name of the training job. Must be between 1 and 63 characters, start with a letter or number, and contain only letters, numbers, and hyphens. */ trainingJobName: pulumi.Input; /** * VPC configuration for the training job. See `vpcConfig` below. */ vpcConfig?: pulumi.Input; } //# sourceMappingURL=trainingJob.d.ts.map