No /opt/ml/input/config/resourceconfig.json error when training with Sagemaker Python SDK locally on WSL

Viewed 518

The goal is to be able to run do Sagemaker local development and train locally with Docker images provided by AWS. I have been able to get this working on a Ubuntu 20.04 VM with code and Docker all running on same VM, but unable to get it working with a WSL Ubuntu 20.04 + Docker Desktop setup.

With the setup below, I have been able to create a simple Docker image that calls a Python script to read from and write data to both WSL directories and the Windows automounted directories to prove the WSL Ubuntu + Docker Desktop is working OK.

Any help is appreciated - been struggling to get this working!

Environment

  • WSL - Ubuntu 20.04 (unable to use WSL 2 due to VPN and DNS issues)
  • Docker Desktop 3.1.0
  • Python 3.7 (pip install numpy pandas sagemaker sagemaker[local])

Setup

[automount]
enabled = true
root = /
options = "metadata,umask=22,fmask=11,case=off"
  • Error occurs when running code on both Windows mount /c or on WSL /home

Issue

The following code is throwing the error when fit() is called.

    mnist_estimator = TensorFlow(entry_point='mnist_tf2.py',
                                 role=dummy_role,
                                 instance_count=1,
                                 instance_type='local',
                                 framework_version='2.2',
                                 source_dir='/home/a632940/dev/sagemaker-local',
                                 py_version='py37',
                                 session=local_session,
                                 distribution={'parameter_server': {'enabled': True}})

    mnist_estimator.fit({'train': training_dataset_path})

Error

Creating network "sagemaker-local" with the default driver
Creating 0j3k45995o-algo-1-prqxb ... done
Attaching to 0j3k45995o-algo-1-prqxb
0j3k45995o-algo-1-prqxb | Reporting training FAILURE
0j3k45995o-algo-1-prqxb | framework error: 
0j3k45995o-algo-1-prqxb | Traceback (most recent call last):
0j3k45995o-algo-1-prqxb |   File "/usr/local/lib/python3.7/site-packages/sagemaker_training/trainer.py", line 66, in train
0j3k45995o-algo-1-prqxb |     env = environment.Environment()
0j3k45995o-algo-1-prqxb |   File "/usr/local/lib/python3.7/site-packages/sagemaker_training/environment.py", line 498, in __init__
0j3k45995o-algo-1-prqxb |     resource_config = resource_config or read_resource_config()
0j3k45995o-algo-1-prqxb |   File "/usr/local/lib/python3.7/site-packages/sagemaker_training/environment.py", line 239, in read_resource_config
0j3k45995o-algo-1-prqxb |     return _read_json(resource_config_file_dir)
0j3k45995o-algo-1-prqxb |   File "/usr/local/lib/python3.7/site-packages/sagemaker_training/environment.py", line 191, in _read_json
0j3k45995o-algo-1-prqxb |     with open(path, "r") as f:
0j3k45995o-algo-1-prqxb | FileNotFoundError: [Errno 2] No such file or directory: '/opt/ml/input/config/resourceconfig.json'
0j3k45995o-algo-1-prqxb | 
0j3k45995o-algo-1-prqxb | [Errno 2] No such file or directory: '/opt/ml/input/config/resourceconfig.json'
0j3k45995o-algo-1-prqxb exited with code 2

Source Code

import os

import boto3
import numpy as np
import sagemaker.session
from sagemaker.local import LocalSession
from sagemaker.tensorflow import TensorFlow

data_files_list = ('train_data.npy', 'train_labels.npy',
                   'eval_data.npy', 'eval_labels.npy')

def download_training_and_eval_data(aws_session):
    if os.path.isfile('./data/train_data.npy') and \
            os.path.isfile('./data/train_labels.npy') and \
            os.path.isfile('./data/eval_data.npy') and \
            os.path.isfile('./data/eval_labels.npy'):
        print('Training and evaluation datasets exist. Skipping Download')
    else:
        print('Downloading training and evaluation dataset')
        s3 = aws_session.resource('s3')
        for filename in data_files_list:
            s3.meta.client.download_file('sagemaker-sample-data-us-east-1', 'tensorflow/mnist/' + filename,
                                         './data/' + filename)


def do_inference_on_local_endpoint(predictor):
    print(f'\nStarting Inference on endpoint.')
    correct_predictions = 0

    train_data = np.load('./data/train_data.npy')
    train_labels = np.load('./data/train_labels.npy')

    predictions = predictor.predict(train_data[:50])
    for i in range(0, 50):
        prediction = np.argmax(predictions['predictions'][i])
        label = train_labels[i]
        print('prediction is {}, label is {}, matched: {}'.format(
            prediction, label, prediction == label))
        if prediction == label:
            correct_predictions = correct_predictions + 1

    print('Calculated Accuracy from predictions: {}'.format(
        correct_predictions / 50))


def main():

    # AWS Setup
    aws_session = boto3.session.Session(profile_name='default')

    download_training_and_eval_data(aws_session)

    local_session = sagemaker.LocalSession()
    local_session.config = {'local': {'local_code': True}}
    dummy_role = 'arn:aws:iam::999999999999:role/Dummy-SageMaker--Role'
    
    training_dataset_path = "file://./data/"

    print('Starting model training.')

    mnist_estimator = TensorFlow(entry_point='mnist_tf2.py',
                                 role=dummy_role,
                                 instance_count=1,
                                 instance_type='local',
                                 framework_version='2.2',
                                 source_dir='/home/a632940/dev/sagemaker-local',
                                 py_version='py37',
                                 session=local_session,
                                 distribution={'parameter_server': {'enabled': True}})

    mnist_estimator.fit({'train': training_dataset_path})
    print('Completed model training')

if __name__ == "__main__":
    main()
0 Answers
Related