Team Ai
Datasetpublic

ysn-rfd/text-dataset-tiny-code-script-py-format

USED of tahamajs/medicine_ds_persian for .parquet file USED of Alijafarixcs2/persian-it-llama2-2k for .parquet file USED of Abirate/english_quotes for .jsonl file NEW FILES (05/12/2025) NEW FILES (12/26/2025) NEW FILES (02/15/2026)

sourceHugging Faceapache-2.0updated 4mo agoView on Hugging Face
3likes1.7kdownloads
day3_2.py108 linesDownload Raw Back to pytorch_study
1import pandas as pd
2from sklearn.model_selection import train_test_split
3from sklearn.preprocessing import LabelEncoder
4from sklearn.feature_extraction.text import CountVectorizer
5import torch
6import torch.nn as nn
7import torch.optim as optim
8from sklearn.metrics import accuracy_score, classification_report
9
10# 1. Load data
11data = pd.read_csv('data.csv')  # Replace 'data.csv' with your dataset
12print("Columns in the dataset:", data.columns)  # Check column names
13
14# 2. Preprocess data
15X = data['text']  # Assuming your text column is named 'text'
16y = data['label']  # Assuming your label column is named 'label'
17
18# Encode labels
19label_encoder = LabelEncoder()
20y_encoded = label_encoder.fit_transform(y)
21
22# Split data into training and testing sets
23X_train, X_test, y_train, y_test = train_test_split(X, y_encoded, test_size=0.2, random_state=42)
24
25# Convert text to numerical features
26vectorizer = CountVectorizer()
27X_train_vectorized = vectorizer.fit_transform(X_train)
28X_test_vectorized = vectorizer.transform(X_test)
29
30# 3. Define the neural network model
31class SentimentModel(nn.Module):
32    def __init__(self, input_size, hidden_size, output_size):
33        super(SentimentModel, self).__init__()
34        self.fc1 = nn.Linear(input_size, hidden_size)
35        self.relu = nn.ReLU()
36        self.fc2 = nn.Linear(hidden_size, output_size)
37        
38    def forward(self, x):
39        x = self.fc1(x)
40        x = self.relu(x)
41        x = self.fc2(x)
42        return x
43
44# Initialize model
45input_size = X_train_vectorized.shape[1]
46hidden_size = 512
47output_size = len(label_encoder.classes_)
48model = SentimentModel(input_size, hidden_size, output_size)
49
50# 4. Define loss function and optimizer
51criterion = nn.CrossEntropyLoss()
52optimizer = optim.Adam(model.parameters(), lr=0.001)
53
54# 5. Train the model
55num_epochs = 1000
56for epoch in range(num_epochs):
57    model.train()
58    optimizer.zero_grad()
59    outputs = model(torch.FloatTensor(X_train_vectorized.toarray()))
60    loss = criterion(outputs, torch.LongTensor(y_train))
61    loss.backward()
62    optimizer.step()
63    print(f'Epoch [{epoch+1}/{num_epochs}], Loss: {loss.item():.4f}')
64
65# 6. Evaluate the model
66model.eval()
67with torch.no_grad():
68    test_outputs = model(torch.FloatTensor(X_test_vectorized.toarray()))
69    _, predicted = torch.max(test_outputs, 1)
70
71    # Calculate accuracy
72    accuracy = accuracy_score(y_test, predicted.numpy())
73    print(f'Accuracy: {accuracy:.4f}')
74    
75    # Detailed classification report
76    print(classification_report(y_test, predicted.numpy(), target_names=label_encoder.classes_))
77
78# 7. Test the model with new sample inputs
79def predict_sentiment(text):
80    # Vectorize the input text
81    text_vectorized = vectorizer.transform([text])
82    with torch.no_grad():
83        output = model(torch.FloatTensor(text_vectorized.toarray()))
84        _, predicted = torch.max(output, 1)
85        return label_encoder.inverse_transform(predicted.numpy())[0]
86
87# Test the model with new sentences
88new_samples = [
89    "It is very good",
90    "Bad",
91    "Good",
92    "loving you",
93    "Loving you",
94    "love you",
95    "Love you",
96    "Very bad",
97    "I love you",
98    "Fuck",
99    "fuck",
100    "bad store",
101    "i dont love this",
102    "not like this"
103]
104
105for sample in new_samples:
106    sentiment = predict_sentiment(sample)
107    print(f'Text: "{sample}" -> Predicted Sentiment: {sentiment}')
108