#BY - VISHNU PRASAD SIVAKUMAR

# a) Time (x-axis)
# b) Average Rating (y-axis) 
# c) Genres (two different graphs on the same axis)

#Opening the IMDb data file for reading
Input = open("data.title.ratings.2018.tsv", "r", encoding = "utf-8")
#Opening the filtered data output file
Output = open("Movies_f.csv", "w", encoding = "utf-8")

#Run through each individual row and coloumn in the input file
for rows in Input:
    #Split each row into columns as a string value which is part of the list called "columns"
    columns = rows.split("\t")
    #Setting a variable called "year" and stating that if a number is listed, then write it into the variable as an integer
    if columns[4] != "\\N":
        year = int(columns[4])

    #Extracting the data for ACTION MOVIES:
        # 1st condition: the 'type' field is the second element (index = 1) and check for rows reading 'Movies'
        # 2nd condition stating that only wirte value out to the Output if it has a rating
        # 3rd condition stating that add only movies which are of the genre 'Action' and genre located at 6th(index) position of the coloums list
        # 4th condition stating that add action movie ratings if the year ('year' variable previously created) it was produced is 2000 or later    
    if columns[1] == "movie" and len(columns[7]) > 0 and columns[6] == "Action" and year >= 2000:
        #if the conditions are met, then write only:
            # the genre of the movie (7th element in the columns-list) and leave a tab space next to the data
            # the year it was produced (5th element in the columns-list) and leave a tab space next to the data
            # the average rating for the responsible movie (8th element in the columns-list) and proceed to the next line 
        Output.write(str(columns[6]) + "\t" + str(columns[4]) + "\t" + str(columns[7]) + "\t" + str(columns[8]) + "\n")
        
    # Same as the previous 'if' statement, but the 3rd condition has been changed into the genre (7th element in the columns-list) 'Comedy'
    if columns[1] == "movie" and len(columns[7]) > 0 and columns[6] == "Comedy" and year >= 2000:
        #Same as previous 'if' statement
        Output.write(str(columns[6]) + "\t" + str(columns[4]) + "\t" + str(columns[7]) + "\t" + str(columns[8]) + "\n")

#Close the filtered data output file
Output.close()
#Close the filtered data output file
Input.close()

#NOTE: Please kindly understand that manual formatting of the data file was required after the editions made through python. Therefore, the final data file submitted was not the table produced by python although it was based on the data obtained from this code. 
