Repository navigation
Expand file tree
/
Copy pathtaskA.py
More file actions
272 lines (216 loc) · 9.51 KB
/
Copy pathtaskA.py
File metadata and controls
272 lines (216 loc) · 9.51 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
"""
Task A: Data Preprocessing
NLP Assignment 2 - Seq2Seq Language Models
This script handles:
1. Dataset loading and validation set extraction
2. Text preprocessing (punctuation removal, ASCII filtering, stopword removal, stemming/lemmatization)
3. Vocabulary creation with 1% frequency threshold
"""
import os
import re
import string
import time
import pandas as pd
import numpy as np
from collections import Counter
from typing import List, Dict, Tuple, Set
import nltk
from nltk.corpus import stopwords
from nltk.stem import PorterStemmer, WordNetLemmatizer
from nltk.tokenize import word_tokenize, sent_tokenize
import spacy
# Download required NLTK data
def download_nltk_data():
"""Download required NLTK data packages"""
packages = ['punkt', 'stopwords', 'wordnet', 'omw-1.4']
for package in packages:
try:
nltk.data.find(f'tokenizers/{package}')
except LookupError:
print(f"Downloading {package}...")
nltk.download(package, quiet=True)
class DataPreprocessor:
"""Handle all data preprocessing tasks"""
def __init__(self):
self.download_dependencies()
self.stop_words = set(stopwords.words('english'))
self.stemmer = PorterStemmer()
self.lemmatizer = WordNetLemmatizer()
# Load spaCy model for sentence tokenization
try:
self.nlp = spacy.load("en_core_web_sm")
except OSError:
print("Downloading spaCy model...")
os.system("python -m spacy download en_core_web_sm")
self.nlp = spacy.load("en_core_web_sm")
def download_dependencies(self):
"""Download required dependencies"""
download_nltk_data()
def remove_punctuation(self, text: str) -> str:
"""Remove punctuation from text"""
return text.translate(str.maketrans('', '', string.punctuation))
def remove_non_ascii(self, text: str) -> str:
"""Remove non-ASCII characters"""
return text.encode('ascii', errors='ignore').decode('ascii')
def remove_stopwords(self, tokens: List[str]) -> List[str]:
"""Remove stopwords from token list"""
return [token for token in tokens if token.lower() not in self.stop_words]
def apply_stemming(self, tokens: List[str]) -> List[str]:
"""Apply stemming to tokens"""
return [self.stemmer.stem(token) for token in tokens]
def apply_lemmatization(self, tokens: List[str]) -> List[str]:
"""Apply lemmatization to tokens"""
return [self.lemmatizer.lemmatize(token) for token in tokens]
def tokenize_text(self, text: str) -> List[str]:
"""Tokenize text into words"""
return word_tokenize(text.lower())
def preprocess_text(self, text: str, use_stemming: bool = False) -> List[str]:
"""
Complete preprocessing pipeline for a single text
Args:
text: Input text to preprocess
use_stemming: Whether to use stemming (False = lemmatization)
Returns:
List of preprocessed tokens
"""
# Remove non-ASCII characters
text = self.remove_non_ascii(text)
# Remove punctuation
text = self.remove_punctuation(text)
# Tokenize
tokens = self.tokenize_text(text)
# Remove stopwords
tokens = self.remove_stopwords(tokens)
# Apply stemming or lemmatization
if use_stemming:
tokens = self.apply_stemming(tokens)
else:
tokens = self.apply_lemmatization(tokens)
# Remove empty tokens and very short tokens
tokens = [token for token in tokens if len(token) > 1]
return tokens
def create_vocabulary(self, tokenized_texts: List[List[str]], min_freq_ratio: float = 0.01) -> Tuple[Dict[str, int], Dict[int, str]]:
"""
Create vocabulary from tokenized texts
Args:
tokenized_texts: List of tokenized texts
min_freq_ratio: Minimum frequency ratio (default 1%)
Returns:
Tuple of (word_to_idx, idx_to_word)
"""
# Count token frequencies
all_tokens = [token for text in tokenized_texts for token in text]
total_tokens = len(all_tokens)
token_counts = Counter(all_tokens)
# Calculate minimum frequency
min_freq = int(total_tokens * min_freq_ratio)
# Filter tokens by minimum frequency
valid_tokens = {token for token, count in token_counts.items() if count >= min_freq}
# Add special tokens
special_tokens = ['<pad>', '<bos>', '<eos>', '<unk>']
vocabulary = special_tokens + sorted(list(valid_tokens))
# Create mappings
word_to_idx = {word: idx for idx, word in enumerate(vocabulary)}
idx_to_word = {idx: word for idx, word in enumerate(vocabulary)}
print(f"Vocabulary size: {len(vocabulary)}")
print(f"Total tokens: {total_tokens}")
print(f"Minimum frequency threshold: {min_freq}")
return word_to_idx, idx_to_word
def texts_to_sequences(self, tokenized_texts: List[List[str]], word_to_idx: Dict[str, int]) -> List[List[int]]:
"""Convert tokenized texts to sequences of indices"""
unk_idx = word_to_idx['<unk>']
return [[word_to_idx.get(token, unk_idx) for token in text] for text in tokenized_texts]
def load_dataset(data_path: str) -> Tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]:
"""
Load and split dataset
Args:
data_path: Path to dataset file
Returns:
Tuple of (train_df, val_df, test_df)
"""
# This is a placeholder - actual implementation depends on dataset format
# For now, we'll create a dummy dataset structure
print(f"Loading dataset from {data_path}")
# TODO: Implement actual dataset loading based on the real dataset format
# This might be CSV, JSON, or another format
# Dummy data for testing
dummy_data = {
'body': [
"This is the first article body. It contains multiple sentences and various words.",
"The second article discusses different topics and has more content.",
"Article number three talks about technology and innovation."
] * 1000, # Repeat to create larger dataset
'title': [
"First Article Title",
"Second Article Title",
"Third Article Title"
] * 1000
}
df = pd.DataFrame(dummy_data)
# Split dataset
train_size = len(df) - 1000 # Reserve 1000 for test
train_df = df[:train_size - 500].copy() # Extract 500 for validation
val_df = df[train_size - 500:train_size].copy()
test_df = df[train_size:].copy()
print(f"Train set size: {len(train_df)}")
print(f"Validation set size: {len(val_df)}")
print(f"Test set size: {len(test_df)}")
return train_df, val_df, test_df
def main():
"""Main preprocessing pipeline"""
start_time = time.time()
print("Starting Task A: Data Preprocessing")
print("=" * 50)
# Initialize preprocessor
preprocessor = DataPreprocessor()
# Load dataset
# TODO: Update with actual dataset path
data_path = "dataset/news_dataset.csv" # Update this path
train_df, val_df, test_df = load_dataset(data_path)
# Preprocess text data
print("\nPreprocessing training data...")
train_bodies = [preprocessor.preprocess_text(body) for body in train_df['body']]
train_titles = [preprocessor.preprocess_text(title) for title in train_df['title']]
print("Preprocessing validation data...")
val_bodies = [preprocessor.preprocess_text(body) for body in val_df['body']]
val_titles = [preprocessor.preprocess_text(title) for title in val_df['title']]
print("Preprocessing test data...")
test_bodies = [preprocessor.preprocess_text(body) for body in test_df['body']]
test_titles = [preprocessor.preprocess_text(title) for title in test_df['title']]
# Create vocabulary from training data
print("\nCreating vocabulary...")
word_to_idx, idx_to_word = preprocessor.create_vocabulary(train_bodies + train_titles)
# Convert to sequences
print("Converting to sequences...")
train_body_seqs = preprocessor.texts_to_sequences(train_bodies, word_to_idx)
train_title_seqs = preprocessor.texts_to_sequences(train_titles, word_to_idx)
val_body_seqs = preprocessor.texts_to_sequences(val_bodies, word_to_idx)
val_title_seqs = preprocessor.texts_to_sequences(val_titles, word_to_idx)
test_body_seqs = preprocessor.texts_to_sequences(test_bodies, word_to_idx)
test_title_seqs = preprocessor.texts_to_sequences(test_titles, word_to_idx)
# Save preprocessed data
processed_data = {
'word_to_idx': word_to_idx,
'idx_to_word': idx_to_word,
'train': {
'bodies': train_body_seqs,
'titles': train_title_seqs
},
'val': {
'bodies': val_body_seqs,
'titles': val_title_seqs
},
'test': {
'bodies': test_body_seqs,
'titles': test_title_seqs
}
}
# Save to pickle file
import pickle
with open('processed_data.pkl', 'wb') as f:
pickle.dump(processed_data, f)
print(f"\nPreprocessing completed in {time.time() - start_time:.2f} seconds")
print("Processed data saved to 'processed_data.pkl'")
return processed_data
if __name__ == "__main__":
main()