-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathshoutLevelScript.py
More file actions
31 lines (23 loc) · 1.21 KB
/
Copy pathshoutLevelScript.py
File metadata and controls
31 lines (23 loc) · 1.21 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
import pandas as pd
# Load the CSV files
csv1 = pd.read_csv('./tidied_dataShoutLevels.csv')
csv2 = pd.read_csv('./processed_data_with_features_2.csv')
# Clean and standardize the 'shout_level' column
csv1['shout_level'] = csv1['shout_level'].str.strip().str.lower()
# Map shout levels to numeric values
shout_mapping = {"shout": "1", "no-shout": "0", "n/a": "1",} # n/a is shout because it's likely i mistakenly put it as it's under the shout
csv1['shout'] = csv1['shout_level'].map(shout_mapping)
# Check for null values in the mapping
if csv1['shout'].isnull().any():
print("Unmapped shout_level values:", csv1.loc[csv1['shout'].isnull(), 'shout_level'].unique())
# Clean the 'file_location' column in both CSVs for matching
csv1['file_location'] = csv1['file_location'].str.strip()
csv2['file_location'] = csv2['file_location'].str.strip()
# Merge CSVs on 'file_location'
csv2 = pd.merge(csv2, csv1[['file_location', 'shout']], on='file_location', how='left')
# Save the updated CSV2
csv2.to_csv('updated_csv2.csv', index=False)
# Check for missing matches
missing_matches = csv2[csv2['shout'].isnull()]
if not missing_matches.empty:
print("Unmatched file_location values in CSV2:", missing_matches['file_location'].unique())