HighFive-OPJ/Exploratory_data_analysis
010
1import pandas as pd
2from sklearn.model_selection import train_test_split
3from collections import Counter
4
5df = pd.read_csv("Movies - Final Annotation - test.csv")
6df.dropna(subset=["sentence", "label"], inplace=True)
7
8corpus_size = len(df)
9test_ratio = 0.25 if 300 <= int(corpus_size * 0.25) <= 500 else 0.20
10print(f"Corpus size: {corpus_size} — Using test ratio: {test_ratio}")
11
12train_df, test_df = train_test_split(df, test_size=test_ratio, random_state=42)
13
14train_df.to_csv("Train_HighFive.csv", index=False, encoding="utf-8")
15test_df.to_csv("Test_HighFive.csv", index=False, encoding="utf-8")
16
17label_counts = Counter(test_df["label"])
18sentence_lengths = test_df["sentence"].apply(lambda s: len(str(s).split()))
19
20report = "# Test Set Statistics\n\n"
21report += f"Total test sentences: {len(test_df)}\n\n"
22
23report += "## Label Distribution\n"
24for label in range(0, 3):
25 report += f"- Label {label}: {label_counts.get(label, 0)}\n"
26
27report += "\n## Sentence Length (in words)\n"
28report += f"- Average: {sentence_lengths.mean():.2f}\n"
29report += f"- Shortest: {sentence_lengths.min()}\n"
30report += f"- Longest: {sentence_lengths.max()}\n"
31
32with open("Dataset.md", "w", encoding="utf-8") as f:
33 f.write(report)
34
35print("Split complete:")
36print("- Train set saved to 'train_set.csv'")
37print("- Test set saved to 'test_set.csv'")
38print("- Test statistics saved to 'dataset.md'")
39 