Skip to content
Shrijayan
Go back

Human Action Recognition from Video

Human Action Recognition from Video

Download original notebook: [Pre Action Recogntion](/notebooks/Pre Action Recogntion)

# Downlaod the UCF50 Dataset
!wget --no-check-certificate https://www.crcv.ucf.edu/data/UCF101/UCF101.rar

#Extract the Dataset
!unrar x UCF101.rar
# Import the required libraries.
!pip install youtube-dl moviepy
!pip install git+https://github.com/TahaAnwar/pafy.git
import os
import cv2
import pafy
import math
import random
import numpy as np
import datetime as dt
import time
import tensorflow as tf
from collections import deque
import matplotlib.pyplot as plt
%matplotlib inline

from moviepy.editor import *


from sklearn.model_selection import train_test_split

import tensorflow as tf
from tensorflow.keras.layers import *
from tensorflow.keras.models import Sequential
from tensorflow.keras.utils import to_categorical
from tensorflow.keras.callbacks import EarlyStopping
from tensorflow.keras.utils import plot_model
!pip install keras-tuner

from tensorflow import keras
from tensorflow.keras.optimizers import Adam
from keras_tuner.tuners import RandomSearch
gpus = tf.config.list_physical_devices('GPU')
if gpus:
  try:
    tf.config.set_visible_devices(gpus[0], 'GPU')
  except RuntimeError as e:
    print(e)

#1. Data Processing

c. Plot at least two samples and their captions (use matplotlib/seaborn/any other library).

# Create a Matplotlib figure and specify the size of the figure.
plt.figure(figsize = (20, 20))

# Get the names of all classes/categories in UCF101.
all_classes_names = os.listdir('/content/UCF-101')

# Generate a list of 2 random values. The values will be between 0-101, 
# where 101 is the total number of class in the dataset. 
random_range = random.sample(range(len(all_classes_names)), 2)

# Iterating through all the generated random values.
for counter, random_index  in enumerate(random_range, 1):

    # Retrieve a Class Name using the Random Index.
    selected_class_Name = all_classes_names[random_index]

    # Retrieve the list of all the video files present in the randomly selected Class Directory.
    video_files_names_list = os.listdir(f'/content/UCF-101/{selected_class_Name}')

    # Randomly select a video file from the list retrieved from the randomly selected Class Directory.
    selected_video_file_name = random.choice(video_files_names_list)

    # Initialize a VideoCapture object to read from the video File.
    video_reader = cv2.VideoCapture(f'/content/UCF-101/{selected_class_Name}/{selected_video_file_name}')
    
    # Read the first frame of the video file.
    _, bgr_frame = video_reader.read()

    # Release the VideoCapture object. 
    video_reader.release()

    # Convert the frame from BGR into RGB format. 
    rgb_frame = cv2.cvtColor(bgr_frame, cv2.COLOR_BGR2RGB)

    # Write the class name on the video frame.
    cv2.putText(rgb_frame, selected_class_Name, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, (255, 255, 255), 2)
    
    # Display the frame.
    plt.subplot(5, 4, counter);plt.imshow(rgb_frame);plt.axis('off')

###a. The candidate can choose a subset of the entire dataset for this task. For Example, 40 classes can be chosen for model training.

# Specify the height and width to which each video frame will be resized in our dataset.
IMAGE_HEIGHT , IMAGE_WIDTH = 64, 64

# Specify the number of frames of a video that will be fed to the model as one sequence.
SEQUENCE_LENGTH = 20

# Specify the directory containing the UCF50 dataset. 
DATASET_DIR = "/content/UCF-101"

# Specify the list containing the names of the classes used for training. Feel free to choose any set of classes.
CLASSES_LIST = ["WalkingWithDog", "TaiChi", "Swing", "HorseRace",'ApplyEyeMakeup','ApplyLipstick','Archery','BabyCrawling','BalanceBeam','BandMarching']
# CLASSES_LIST = ["WalkingWithDog", "TaiChi", "Swing", "HorseRace"]

###b. Convert the data into the correct format which could be used for the DL model.

def frames_extraction(video_path):
    '''
    This function will extract the required frames from a video after resizing and normalizing them.
    Args:
        video_path: The path of the video in the disk, whose frames are to be extracted.
    Returns:
        frames_list: A list containing the resized and normalized frames of the video.
    '''

    # Declare a list to store video frames.
    frames_list = []
    
    # Read the Video File using the VideoCapture object.
    video_reader = cv2.VideoCapture(video_path)

    # Get the total number of frames in the video.
    video_frames_count = int(video_reader.get(cv2.CAP_PROP_FRAME_COUNT))

    # Calculate the the interval after which frames will be added to the list.
    skip_frames_window = max(int(video_frames_count/SEQUENCE_LENGTH), 1)

    # Iterate through the Video Frames.
    for frame_counter in range(SEQUENCE_LENGTH):

        # Set the current frame position of the video.
        video_reader.set(cv2.CAP_PROP_POS_FRAMES, frame_counter * skip_frames_window)

        # Reading the frame from the video. 
        success, frame = video_reader.read() 

        # Check if Video frame is not successfully read then break the loop
        if not success:
            break

        # Resize the Frame to fixed height and width.
        resized_frame = cv2.resize(frame, (IMAGE_HEIGHT, IMAGE_WIDTH))
        
        # Normalize the resized frame by dividing it with 255 so that each pixel value then lies between 0 and 1
        normalized_frame = resized_frame / 255
        
        # Append the normalized frame into the frames list
        frames_list.append(normalized_frame)
    
    # Release the VideoCapture object. 
    video_reader.release()

    # Return the frames list.
    return frames_list
def create_dataset():

    # Declared Empty Lists to store the features, labels and video file path values.
    features = []
    labels = []
    video_files_paths = []
    
    # Iterating through all the classes mentioned in the classes list
    for class_index, class_name in enumerate(CLASSES_LIST):
        
        # Display the name of the class whose data is being extracted.
        print(f'Extracting Data of Class: {class_name}')
        
        # Get the list of video files present in the specific class name directory.
        files_list = os.listdir(os.path.join(DATASET_DIR, class_name))
        
        # Iterate through all the files present in the files list.
        for file_name in files_list:
            
            # Get the complete video path.
            video_file_path = os.path.join(DATASET_DIR, class_name, file_name)

            # Extract the frames of the video file.
            frames = frames_extraction(video_file_path)

            # Check if the extracted frames are equal to the SEQUENCE_LENGTH specified above.
            # So ignore the vides having frames less than the SEQUENCE_LENGTH.
            if len(frames) == SEQUENCE_LENGTH:

                # Append the data to their repective lists.
                features.append(frames)
                labels.append(class_index)
                video_files_paths.append(video_file_path)

    # Converting the list to numpy arrays
    features = np.asarray(features)
    labels = np.array(labels)  
    
    # Return the frames, class index, and video file path.
    return features, labels, video_files_paths
# Create the dataset.
features, labels, video_files_paths = create_dataset()
# Using Keras's to_categorical method to convert labels into one-hot-encoded vectors
one_hot_encoded_labels = to_categorical(labels)

c. Plot at least two samples and their captions (use matplotlib/seaborn/any other library).

# Plot samples
fig, axes = plt.subplots(1, 2, figsize=(10, 5))
axes[0].imshow(features[10][0])
axes[0].set_title(labels[0])
axes[1].imshow(features[1][0])
axes[1].set_title(labels[1])
plt.show()

##d. Load the data into train and test data in the required format.

seed_constant = 27
np.random.seed(seed_constant)
random.seed(seed_constant)
tf.random.set_seed(seed_constant)
# Split the Data into Train ( 75% ) and Test Set ( 25% ).
features_train, features_test, labels_train, labels_test = train_test_split(features, one_hot_encoded_labels,
                                                                            test_size = 0.25, shuffle = True,
                                                                            random_state = seed_constant)

#2. Model Building ##a. Use any pretrained model or a custom-built model as CNN encoder for feature extraction.

##b. Create k-layered LSTM model and other relevant layers.

##c. Add one layer of dropout at the appropriate position and give reasons #model h2

!rm -r my_dir/
def build_model(hp):
    cnn_model = Sequential()
    cnn_model.add(Conv2D(filters=hp.Int('filters_1', 8, 64, step=8), kernel_size=(hp.Choice('kernel_height_1', [3, 5]), hp.Choice('kernel_width_1', [3, 5])), activation='relu', input_shape=(IMAGE_HEIGHT, IMAGE_WIDTH, 3)))
    cnn_model.add(MaxPooling2D(pool_size=(2,2)))

    cnn_model.add(Conv2D(filters=hp.Int('filters_2', 8, 64, step=8), kernel_size=(hp.Choice('kernel_height_2', [3, 5]), hp.Choice('kernel_width_2', [3, 5])), activation='relu'))
    cnn_model.add(MaxPooling2D(pool_size=(2,2)))

    cnn_model.add(Conv2D(filters=hp.Int('filters_3', 8, 64, step=8), kernel_size=(hp.Choice('kernel_height_3', [3, 5]), hp.Choice('kernel_width_3', [3, 5])), activation='relu'))
    cnn_model.add(MaxPooling2D(pool_size=(2,2)))

    cnn_model.add(Flatten())
    cnn_output_shape = cnn_model.layers[-1].output_shape[1:]

    lstm_model = Sequential()
    lstm_model.add(LSTM(units=hp.Int('lstm_units_1', 32, 128, step=32), input_shape=(SEQUENCE_LENGTH, np.prod(cnn_output_shape)), return_sequences=True))
    lstm_model.add(Dropout(hp.Float('dropout_rate_1', 0.2, 0.5, step=0.1)))

    lstm_model.add(LSTM(units=hp.Int('lstm_units_2', 32, 128, step=32), return_sequences=True))
    lstm_model.add(Dropout(hp.Float('dropout_rate_2', 0.2, 0.5, step=0.1)))

    lstm_model.add(LSTM(units=hp.Int('lstm_units_3', 32, 128, step=32), return_sequences=False))

    lstm_output_shape = lstm_model.layers[-1].output_shape[-1]

    model = Sequential()
    model.add(TimeDistributed(cnn_model, input_shape=(SEQUENCE_LENGTH, IMAGE_HEIGHT, IMAGE_WIDTH, 3)))
    model.add(TimeDistributed(Flatten()))
    model.add(lstm_model)
    model.add(Dense(len(CLASSES_LIST), activation='softmax'))

    model.compile(loss='categorical_crossentropy', optimizer=Adam(learning_rate=hp.Float('learning_rate', 0.001, 0.01, step=0.001)), metrics=['accuracy'])
    return model

# Create a Keras Tuner RandomSearch instance
tuner = RandomSearch(
    build_model, objective='val_accuracy', 
    max_trials=10, 
    executions_per_trial=1, 
    directory='my_dir', 
    project_name='convlstm_tuning')
# Perform hyperparameter tuning
tuner.search_space_summary()
tuner.search(features_train, labels_train, epochs=10, validation_data=(features_test, labels_test))
# Retrieve the best model
best_model = tuner.get_best_models(num_models=1)[0]
best_hyperparameters = tuner.get_best_hyperparameters(num_trials=1)[0]
# Print the best model summary
best_model.summary()

#4. Model Training: ##a. Train the model for an appropriate number of epochs.

# Train the best model and store the history
start_time = time.time()
history = best_model.fit(features_train, labels_train, epochs=50, batch_size=4, shuffle=True, validation_split=0.2, callbacks=[EarlyStopping(monitor='val_loss', patience=10, mode='min', restore_best_weights=True)])
end_time = time.time()

##b. Print the train and validation loss for each epoch. Use the appropriate batch size.

# Print the train and validation loss for each epoch
for epoch, (train_loss, val_loss) in enumerate(zip(history.history['loss'], history.history['val_loss'])):
    print(f"Epoch {epoch + 1}: Train Loss = {train_loss}, Validation Loss = {val_loss}")

##c. Plot the loss and accuracy history graphs for both train and validation set.

# Plot the loss and accuracy history graphs for both train and validation set
plt.figure(figsize=(12, 4))

plt.subplot(1, 2, 1)
plt.plot(history.history['loss'], label='Train Loss')
plt.plot(history.history['val_loss'], label='Validation Loss')
plt.xlabel('Epochs')
plt.ylabel('Loss')
plt.legend()

plt.subplot(1, 2, 2)
plt.plot(history.history['accuracy'], label='Train Accuracy')
plt.plot(history.history['val_accuracy'], label='Validation Accuracy')
plt.xlabel('Epochs')
plt.ylabel('Accuracy')
plt.legend()

plt.show()
# Evaluate the best model
model_evaluation_history = best_model.evaluate(features_test, labels_test)

##d. Print the total time taken for training.

# Print the total time taken for training
print(f"Total time taken for training: {end_time - start_time} seconds")

##a. Take 5 random data from the test set and perform Expression recognition.

from sklearn.metrics import confusion_matrix, classification_report

# Get 5 random samples from the test set
random_indices = np.random.choice(len(features_test), size=5, replace=False)
random_features = features_test[random_indices]
random_labels = labels_test[random_indices]
np.argmax(random_labels, axis=1)
# Use the best model to predict the labels for the random samples
predicted_labels = best_model.predict(random_features)
predicted_labels = np.argmax(predicted_labels, axis=1)
predicted_labels

##b. Print confusion metrics and classification report for the test data.

# Print the confusion matrix and classification report for the random samples
print("Confusion Matrix:")
print(confusion_matrix(np.argmax(random_labels, axis=1), predicted_labels))
print("\nClassification Report:")
print(classification_report(np.argmax(random_labels, axis=1), predicted_labels, zero_division=1))

v1

# Perform hyperparameter tuning
tuner.search_space_summary()
tuner.search(features_train, labels_train, epochs=3, batch_size=4, shuffle=True, validation_split=0.2, callbacks=[EarlyStopping(monitor='val_loss', patience=10, mode='min', restore_best_weights=True)])
# Retrieve the best model
best_model = tuner.get_best_models(num_models=1)[0]
best_hyperparameters = tuner.get_best_hyperparameters(num_trials=1)[0]
# Evaluate the best model
model_evaluation_history = best_model.evaluate(features_test, labels_test)

model h1 error

# Define the function to create the model with hyperparameters to be tuned
def create_model(hp):
    cnn_model = Sequential()
    cnn_model.add(Conv2D(filters=hp.Int('filters1', min_value=16, max_value=64, step=16), kernel_size=(3,3), activation='relu', input_shape=(IMAGE_HEIGHT, IMAGE_WIDTH, 3)))
    cnn_model.add(MaxPooling2D(pool_size=(2,2)))

    cnn_model.add(Conv2D(filters=hp.Int('filters2', min_value=32, max_value=128, step=32), kernel_size=(3,3), activation='relu'))
    cnn_model.add(MaxPooling2D(pool_size=(2,2)))

    cnn_model.add(Flatten())
    cnn_output_shape = cnn_model.layers[-1].output_shape[1:]

    lstm_model = Sequential()
    lstm_model.add(LSTM(units=hp.Int('units1', min_value=32, max_value=128, step=32), input_shape=(SEQUENCE_LENGTH, np.prod(cnn_output_shape)), return_sequences=True))
    lstm_model.add(Dropout(hp.Float('dropout1', min_value=0.0, max_value=0.5, step=0.1)))

    lstm_model.add(LSTM(units=hp.Int('units2', min_value=16, max_value=64, step=16), return_sequences=True))
    lstm_model.add(Dropout(hp.Float('dropout2', min_value=0.0, max_value=0.5, step=0.1)))

    lstm_model.add(LSTM(units=hp.Int('units3', min_value=8, max_value=32, step=8), return_sequences=False))

    lstm_output_shape = lstm_model.layers[-1].output_shape[-1]

    model = Sequential()
    model.add(TimeDistributed(cnn_model, input_shape=(SEQUENCE_LENGTH, IMAGE_HEIGHT, IMAGE_WIDTH, 3)))
    model.add(TimeDistributed(Flatten()))
    model.add(lstm_model)
    model.add(Dense(len(CLASSES_LIST), activation='softmax'))

    model.compile(optimizer=Adam(learning_rate=hp.Choice('learning_rate', values=[1e-2, 1e-3, 1e-4])), loss='categorical_crossentropy', metrics=['accuracy'])

    return model
# Define the parameter search space
tuner = RandomSearch(
    create_model,
    objective='val_accuracy',
    max_trials=10,
    directory='my_dir',
    project_name='cnn_lstm_hyperparams')
print(features_train.shape)
print(features_test.shape)
print(labels_train.shape)
print(labels_test.shape)
# Train the model using the hyperparameters obtained from tuner
tuner.search(features_train, labels_train, epochs=2, validation_data=(features_test, labels_test))
# Get the best model and hyperparameters
best_model = tuner.get_best_models(num_models=1)[0]
best_hyperparams = tuner.get_best_hyperparameters(num_trials=1)[0]
# Display the model's summary
cnn_model.summary()
lstm_model.summary()
model.summary()

default model

def create_convlstm_model():

    # Define the CNN model architecture
    cnn_model = Sequential()
    cnn_model.add(Conv2D(filters=16, kernel_size=(3,3), activation='relu', input_shape=(IMAGE_HEIGHT, IMAGE_WIDTH, 3)))
    cnn_model.add(MaxPooling2D(pool_size=(2,2)))

    cnn_model.add(Conv2D(filters=32, kernel_size=(3,3), activation='relu'))
    cnn_model.add(MaxPooling2D(pool_size=(2,2)))

    cnn_model.add(Conv2D(filters=64, kernel_size=(3,3), activation='relu'))
    cnn_model.add(MaxPooling2D(pool_size=(2,2)))

    cnn_model.add(Flatten())
    cnn_output_shape = cnn_model.layers[-1].output_shape[1:]  # shape of CNN output
    
    # Define the LSTM model architecture
    lstm_model = Sequential()
    lstm_model.add(LSTM(units=64, input_shape=(SEQUENCE_LENGTH, np.prod(cnn_output_shape)), return_sequences=True))
    lstm_model.add(Dropout(0.2))

    lstm_model.add(LSTM(units=32, return_sequences=True))
    lstm_model.add(Dropout(0.2))

    lstm_model.add(LSTM(units=16, return_sequences=False))

    lstm_output_shape = lstm_model.layers[-1].output_shape[-1]  # shape of LSTM output
    
    # Combine the CNN and LSTM models
    model = Sequential()
    model.add(TimeDistributed(cnn_model, input_shape=(SEQUENCE_LENGTH, IMAGE_HEIGHT, IMAGE_WIDTH, 3)))
    model.add(TimeDistributed(Flatten()))
    model.add(lstm_model)
    model.add(Dense(len(CLASSES_LIST), activation='softmax'))
    
    # Display the model's summary
    cnn_model.summary()
    lstm_model.summary()
    model.summary()
    
    
    # Return the constructed CNN-LSTM model
    return model

##e. Print the model summary

# Construct the required convlstm model.
convlstm_model = create_convlstm_model()

# Display the success message. 
print("Model Created Successfully!")
# Plot the structure of the contructed model.
plot_model(convlstm_model, to_file = 'convlstm_model_structure_plot.png', show_shapes = True, show_layer_names = True)

#3. Model Compilation ##a. Compile the model with the appropriate loss function

##b. Use an appropriate optimizer

# Create an Instance of Early Stopping Callback
early_stopping_callback = EarlyStopping(monitor = 'val_loss', patience = 10, mode = 'min', restore_best_weights = True)

# Compile the model and specify loss function, optimizer and metrics values to the model
convlstm_model.compile(loss = 'categorical_crossentropy', optimizer = 'Adam', metrics = ["accuracy"])

#4. Model Training: ##a. Train the model for an appropriate number of epochs.

start_time = time.time()
# Start training the model.
convlstm_model_training_history = convlstm_model.fit(x = features_train, y = labels_train, epochs = 50, batch_size = 4,
                                                    shuffle = True, validation_split = 0.2, 
                                                    callbacks = [early_stopping_callback])
end_time = time.time()

##b. Print the train and validation loss for each epoch. Use the appropriate batch size.

# Evaluate the trained model.
model_evaluation_history = convlstm_model.evaluate(features_test, labels_test)

##c. Plot the loss and accuracy history graphs for both train and validation set.

def plot_metric(model_training_history, metric_name_1, metric_name_2, plot_name):

    # Get metric values using metric names as identifiers.
    metric_value_1 = model_training_history.history[metric_name_1]
    metric_value_2 = model_training_history.history[metric_name_2]
    
    # Construct a range object which will be used as x-axis (horizontal plane) of the graph.
    epochs = range(len(metric_value_1))

    # Plot the Graph.
    plt.plot(epochs, metric_value_1, 'blue', label = metric_name_1)
    plt.plot(epochs, metric_value_2, 'red', label = metric_name_2)

    # Add title to the plot.
    plt.title(str(plot_name))

    # Add legend to the plot.
    plt.legend()
# Visualize the training and validation accuracy metrices.
plot_metric(best_model, 'accuracy', 'val_accuracy', 'Total Accuracy vs Total Validation Accuracy') 
# Visualize the training and validation loss metrices.
plot_metric(convlstm_model_training_history, 'loss', 'val_loss', 'Total Loss vs Total Validation Loss')

##d. Print the total time taken for training.

# Calculate elapsed time for training in minutes or seconds.
elapsed_time_sec = end_time - start_time
elapsed_time_min = elapsed_time_sec / 60.0

print(f"Total training time: {elapsed_time_min:.2f} minutes ({elapsed_time_sec:.2f} seconds)")

5. Model Evaluation

a. Take 5 random data from the test set and perform Expression recognition.

b. Print confusion metrics and classification report for the test data.

def predict_on_video(video_file_path, output_file_path, SEQUENCE_LENGTH):
    '''
    This function will perform action recognition on a video using the LRCN model.
    Args:
    video_file_path:  The path of the video stored in the disk on which the action recognition is to be performed.
    output_file_path: The path where the ouput video with the predicted action being performed overlayed will be stored.
    SEQUENCE_LENGTH:  The fixed number of frames of a video that can be passed to the model as one sequence.
    '''

    # Initialize the VideoCapture object to read from the video file.
    video_reader = cv2.VideoCapture(video_file_path)

    # Get the width and height of the video.
    original_video_width = int(video_reader.get(cv2.CAP_PROP_FRAME_WIDTH))
    original_video_height = int(video_reader.get(cv2.CAP_PROP_FRAME_HEIGHT))

    # Initialize the VideoWriter Object to store the output video in the disk.
    video_writer = cv2.VideoWriter(output_file_path, cv2.VideoWriter_fourcc('M', 'P', '4', 'V'), 
                                   video_reader.get(cv2.CAP_PROP_FPS), (original_video_width, original_video_height))

    # Declare a queue to store video frames.
    frames_queue = deque(maxlen = SEQUENCE_LENGTH)

    # Initialize a variable to store the predicted action being performed in the video.
    predicted_class_name = ''

    # Iterate until the video is accessed successfully.
    while video_reader.isOpened():

        # Read the frame.
        ok, frame = video_reader.read() 
        
        # Check if frame is not read properly then break the loop.
        if not ok:
            break

        # Resize the Frame to fixed Dimensions.
        resized_frame = cv2.resize(frame, (IMAGE_HEIGHT, IMAGE_WIDTH))
        
        # Normalize the resized frame by dividing it with 255 so that each pixel value then lies between 0 and 1.
        normalized_frame = resized_frame / 255

        # Appending the pre-processed frame into the frames list.
        frames_queue.append(normalized_frame)

        # Check if the number of frames in the queue are equal to the fixed sequence length.
        if len(frames_queue) == SEQUENCE_LENGTH:

            # Pass the normalized frames to the model and get the predicted probabilities.
            predicted_labels_probabilities = convlstm_model.predict(np.expand_dims(frames_queue, axis = 0))[0]

            # Get the index of class with highest probability.
            predicted_label = np.argmax(predicted_labels_probabilities)

            # Get the class name using the retrieved index.
            predicted_class_name = CLASSES_LIST[predicted_label]

        # Write predicted class name on top of the frame.
        cv2.putText(frame, predicted_class_name, (10, 30), cv2.FONT_HERSHEY_SIMPLEX, 1, (0, 255, 0), 2)

        # Write The frame into the disk using the VideoWriter Object.
        video_writer.write(frame)
        
    # Release the VideoCapture and VideoWriter objects.
    video_reader.release()
    video_writer.release()
# Construct the output video path.
output_video_file_path = '/content/test.avi'

# Perform Action Recognition on the Test Video.
predict_on_video('/content/UCF-101/WalkingWithDog/v_WalkingWithDog_g02_c04.avi', output_video_file_path, SEQUENCE_LENGTH)

# Display the output video.
VideoFileClip(output_video_file_path, audio=False, target_resolution=(300,None)).ipython_display()
# Construct the output video path.
output_video_file_path = '/content/test1.avi'

# Perform Action Recognition on the Test Video.
predict_on_video('/content/UCF-101/Swing/v_Swing_g07_c04.avi', output_video_file_path, SEQUENCE_LENGTH)

# Display the output video.
VideoFileClip(output_video_file_path, audio=False, target_resolution=(300,None)).ipython_display()

Share this post on:

Previous Post
Question Generation with Transformers
Next Post
Voice-Controlled Computer Vision Assistant (VisionA)