-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathdev_process.py
More file actions
119 lines (92 loc) · 3.03 KB
/
Copy pathdev_process.py
File metadata and controls
119 lines (92 loc) · 3.03 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
import pandas as pd
import os
BASE_PATH = "E:/pep/data/dev"
OUTPUT_PATH = "E:/pep/dev/dev_final.csv"
def process_tamil():
path = f"{BASE_PATH}/tamil/dev.csv"
img_folder = f"{BASE_PATH}/tamil"
df = pd.read_csv(path)
# cols: image_id, transcriptions, original_labels, irish_labels, chinese_labels
df = df.rename(columns={
"transcriptions": "transcription",
"original_labels": "india_label",
"irish_labels": "western_label",
"chinese_labels": "china_label"
})
return finalize(df, img_folder)
def process_malayalam():
path = f"{BASE_PATH}/malayalam/dev.csv"
img_folder = f"{BASE_PATH}/malayalam"
df = pd.read_csv(path)
# cols: image_id, transcriptions, original_labels, irish_labels, chinese_labels
df = df.rename(columns={
"transcriptions": "transcription",
"original_labels": "india_label",
"irish_labels": "western_label",
"chinese_labels": "china_label"
})
return finalize(df, img_folder)
def process_chinese():
path = f"{BASE_PATH}/chinese/dev.csv"
img_folder = f"{BASE_PATH}/chinese"
df = pd.read_csv(path)
# cols: image_id, transcriptions, original_labels, indian_labels, irish_labels
df = df.rename(columns={
"transcriptions": "transcription",
"original_labels": "china_label",
"indian_labels": "india_label",
"irish_labels": "western_label"
})
return finalize(df, img_folder)
def process_western():
path = f"{BASE_PATH}/western/dev.csv"
img_folder = f"{BASE_PATH}/western"
df = pd.read_csv(path)
# cols: image_id, transcriptions, indian_labels, chinese_labels
df = df.rename(columns={
"transcriptions": "transcription",
"indian_labels": "india_label",
"chinese_labels": "china_label"
})
df["western_label"] = None
return finalize(df, img_folder)
def finalize(df, img_folder):
label_map = {
"misogyny": 1,
"not-misogyny": 0
}
for col in ["india_label", "western_label", "china_label"]:
if col in df.columns:
df[col] = df[col].map(label_map)
df["image_path"] = df["image_id"].apply(
lambda x: os.path.join(img_folder, f"{x}.jpg")
)
return df[[
"image_id",
"transcription",
"india_label",
"western_label",
"china_label",
"image_path"
]]
# ===== RUN =====
dfs = [
process_tamil(),
process_malayalam(),
process_chinese(),
process_western()
]
dev_final = pd.concat(dfs, ignore_index=True)
# ===== VALIDATION =====
print("Total rows:", len(dev_final))
print("Missing values:")
print(dev_final.isnull().sum())
print()
for col in ["india_label", "western_label", "china_label"]:
v = dev_final[col].notna().sum()
p = (dev_final[col] == 1).sum()
n = (dev_final[col] == 0).sum()
print(f" {col}: valid={v}, 1={p}, 0={n}, NaN={dev_final[col].isna().sum()}")
os.makedirs(os.path.dirname(OUTPUT_PATH), exist_ok=True)
dev_final.to_csv(OUTPUT_PATH, index=False)
print(f"\nSaved: {OUTPUT_PATH}")