-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy path1_preprocess_data.py
More file actions
91 lines (73 loc) · 2.78 KB
/
Copy path1_preprocess_data.py
File metadata and controls
91 lines (73 loc) · 2.78 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
import json
import re
import pandas as pd
# Load the CSV file with keep_default_na=False to prevent "NA" from being treated as NaN
df = pd.read_csv("data/persian.csv", keep_default_na=False, na_values=[""])
# Define valid labels
labels_structure = {
"MT": [],
"LY": [],
"SP": ["it"],
"ID": [],
"NA": ["ne", "sr", "nb"],
"HI": ["re"],
"IN": ["en", "ra", "dtp", "fi", "lt"],
"OP": ["rv", "ob", "rs", "av"],
"IP": ["ds", "ed"],
}
# Create a set of all valid labels (main + sub)
valid_labels = set(labels_structure.keys())
for sub_labels in labels_structure.values():
valid_labels.update(sub_labels)
# Function to clean and consolidate labels
def consolidate_labels(group):
# Collect all unique values from Turku_NLP and Turku_NLP_sub
all_labels = set()
for val in group["Turku_NLP"]:
if val and val != "": # Check for empty strings
# Split by semicolon and clean
labels = [label.strip() for label in str(val).split(";")]
all_labels.update(labels)
for val in group["Turku_NLP_sub"]:
if val and val != "": # Check for empty strings
# Split by semicolon and clean
labels = [label.strip() for label in str(val).split(";")]
all_labels.update(labels)
# Remove non-alphabetic characters (keep only letters and spaces), then strip and filter empties
all_labels = [re.sub(r"[^a-zA-Z\s]", "", label).strip() for label in all_labels]
all_labels = [label for label in all_labels if label]
# Filter to keep only valid labels
all_labels = [label for label in all_labels if label in valid_labels]
# Sort alphabetically
all_labels = sorted(all_labels)
# Join with spaces
return " ".join(all_labels)
# Select only the relevant columns and drop duplicates based on 'id'
consolidated = (
df.groupby("id")
.agg(
{
"text": "first", # Take the first text (they should all be the same)
}
)
.reset_index()
)
# Add the consolidated label column
consolidated["label"] = df.groupby("id").apply(consolidate_labels).values
# Create list of dictionaries for JSONL
records = consolidated.to_dict("records")
# Save to JSONL
output_file = "data/persian_consolidated.jsonl"
with open(output_file, "w", encoding="utf-8") as f:
for record in records:
json.dump(record, f, ensure_ascii=False)
f.write("\n")
print(f"✓ Consolidated {len(df)} rows into {len(consolidated)} unique documents")
print(f"✓ Saved to {output_file}")
print(f"\nValid labels: {sorted(valid_labels)}")
print(f"\nSample of first 3 records:")
for i, record in enumerate(records[:3]):
print(f"\nRecord {i + 1}:")
print(f" ID: {record['id']}")
print(f" Label: {record['label']}")
print(f" Text preview: {record['text'][:100]}...")