import os
import pandas as pd
import argparse

# Argument parser to get the file prefix from the command line
parser = argparse.ArgumentParser(description='Merge CSV files and find top 5 max volumes before 12:00 PM.')
parser.add_argument('file_prefix', type=str, help='The prefix of the CSV files to process')
args = parser.parse_args()

# Set the directory containing the CSV files
directory = '/var/www/html/clientData/RA1383/input'
file_prefix = args.file_prefix

# Initialize an empty DataFrame to store the merged data
merged_df = pd.DataFrame()

# Loop through all files in the directory
for filename in os.listdir(directory):
    if filename.startswith(file_prefix) and filename.endswith('.csv'):
        file_path = os.path.join(directory, filename)
        
        # Read the CSV file into a DataFrame
        df = pd.read_csv(file_path)
        
        # Convert the 'time' column to datetime format
        df['time'] = pd.to_datetime(df['time'], format='%d-%m-%Y %H:%M:%S')
        
        # Filter the DataFrame to keep only rows before 12:00 PM
        df_filtered = df[df['time'].dt.time <= pd.to_datetime('12:00:00').time()]
        
        # Add a column to track the filename for each row
        df_filtered['filename'] = filename
        
        # Append the filtered data to the merged DataFrame
        merged_df = pd.concat([merged_df, df_filtered])

# Find the top 5 maximum volumes ('v') in the merged DataFrame
top_5_max_volumes = merged_df.nlargest(20, 'intv')

# Output the top 5 maximum volumes along with the source filenames
print("Top 10 maximum volumes before 12:00 PM:")
print(top_5_max_volumes[['filename', 'time', 'intv']])
