PyTorch torchtext
Although the PyTorch ecosystem hastorchvisionprocessing image data,torchaudioprocessing audio data, but in the field of text processing, the officially providedtorchtextlibrary has undergone some changes. This section introduces how to perform text data preprocessing, vocabulary building, data loading, and other operations using various methods.
Note: The torchtext library has undergone some refactoring. It is recommended to usetorchtext.legacyor build your own text processing pipeline. The latest torchtext version has returned and provides a more modern API.
1. Text Data Preprocessing Basics
Text preprocessing is the first step in NLP tasks, including tokenization, vocabulary building, encoding, and other operations.
1.1 Basic Text Processing Pipeline
Example
from collections import Counter
class SimpleTokenizer:
"""
Simple tokenizer: tokenize by spaces and punctuation
"""
def __init__(self):
# Punctuation mapping
self.punctuation = str.maketrans('', '', '.,!?;:"\'-()[]{}')
def tokenize(self, text):
# Convert to lowercase
text = text.lower()
# Remove punctuation
text = text.translate(self.punctuation)
# Tokenize
tokens = text.split()
return tokens
class Vocabulary:
"""
Vocabulary building
"""
def __init__(self, min_freq=2, max_size=10000):
self.min_freq = min_freq
self.max_size = max_size
self.word2idx = {'<PAD>': 0, '<UNK>': 1}
self.idx2word = {0: '<PAD>', 1: '<UNK>'}
self.word_count = Counter()
def build_vocab(self, texts):
"""Build vocabulary from a list of texts"""
tokenizer = SimpleTokenizer()
# Count word frequencies
for text in texts:
tokens = tokenizer.tokenize(text)
self.word_count.update(tokens)
# Build vocabulary
for word, count in self.word_count.most_common(self.max_size):
if count < self.min_freq:
break
if word not in self.word2idx:
idx = len(self.word2idx)
self.word2idx[word] = idx
self.idx2word[idx] = word
print(f"Vocabulary size: {len(self.word2idx)}")
def encode(self, text, max_len=50):
"""Encode text as a sequence of indices"""
tokenizer = SimpleTokenizer()
tokens = tokenizer.tokenize(text)[:max_len]
# Encode
indices = [
self.word2idx.get(token, self.word2idx['<UNK>'])
for token in tokens
]
# Pad with zeros
if len(indices) < max_len:
indices += [self.word2idx['<PAD>']] * (max_len - len(indices))
return indices
def decode(self, indices):
"""Decode a sequence of indices into text"""
tokens = [self.idx2word.get(idx, '<UNK>') for idx in indices]
return ' '.join(tokens)
# Usage example
texts = [
"Hello world",
"This is a test",
"PyTorch is great for deep learning",
"Natural language processing is fun",
"Deep learning enables many applications",
]
vocab = Vocabulary(min_freq=1, max_size=100)
vocab.build_vocab(texts)
# Encode
encoded = vocab.encode("Hello deep learning world!", max_len=10)
print(f"Encoded result: {encoded}")
# Decode
decoded = vocab.decode(encoded)
print(f"Decoded result: {decoded}")
2. Using torchtext (Legacy)
If torchtext is installed, you can use the legacy module for text processing.
2.1 Installation and Import
Example
# pip install torchtext
# Import the legacy module
try:
import torchtext
from torchtext.legacy import data
from torchtext.legacy import datasets
print(f"torchtext version: {torchtext.__version__}")
TORCHTEXT_AVAILABLE = True
except ImportError:
print("torchtext is not installed, using custom implementation")
TORCHTEXT_AVAILABLE = False
# Or try the new version
try:
from torchtext import transforms
from torchtext.datasets import AG_NEWS
print("New version of torchtext is available")
except ImportError:
print("New version of torchtext is not available")
2.2 Field and Dataset
Example
from torchtext.legacy import data
from torchtext.legacy import datasets
from torchtext.legacy.data import Field, TabularDataset, BucketIterator
# Define fields
TEXT = Field(
tokenize='spacy', # Use spacy for tokenization
tokenizer_language='en_core_web_sm',
lower=True, # Convert to lowercase
include_lengths=True, # Return sequence length
batch_first=True # Batch dimension first
)
LABEL = Field(
sequential=False,
use_vocab=False,
dtype=torch.long
)
# Define dataset
# Assume CSV file format: label,text
# 1,This is a positive review
# 0,This is a negative review
# Or use a built-in dataset (e.g., IMDB)
# train_data, test_data = datasets.IMDB.splits(TEXT, LABEL)
print("Field and Dataset configuration complete")
2.3 Vocabulary Building and Iterators
Example
# TEXT.build_vocab(train_data, vectors="glove.6B.100d", max_size=10000)
# Create iterator
# train_iterator, test_iterator = BucketIterator.splits(
# (train_data, test_data),
# batch_size=32,
# sort_within_batch=True,
# device=torch.device('cuda')
# )
# Iterate training
# for batch in train_iterator:
# text, lengths = batch.text
# labels = batch.label
# # Training code
print("Vocabulary and iterator configuration complete")
3. Custom NLP Data Pipeline
It is recommended to use a more flexible custom data processing approach that does not depend on torchtext.
3.1 Dataset Class Implementation
Example
from torch.utils.data import Dataset, DataLoader
from collections import Counter
import re
class TextDataset(Dataset):
"""
Text dataset
"""
def __init__(self, texts, labels, vocab=None, max_len=128, min_freq=2):
self.texts = texts
self.labels = labels
self.max_len = max_len
# Build or use an existing vocabulary
if vocab is None:
self.vocab = self._build_vocab(texts, min_freq)
else:
self.vocab = vocab
def _build_vocab(self, texts, min_freq):
"""Build vocabulary"""
word_counts = Counter()
for text in texts:
tokens = self._tokenize(text)
word_counts.update(tokens)
# Create vocabulary
word2idx = {'<PAD>': 0, '<UNK>': 1}
for word, count in word_counts.items():
if count >= min_freq:
word2idx[word] = len(word2idx)
return word2idx
def _tokenize(self, text):
"""Simple tokenization"""
# Convert to lowercase
text = text.lower()
# Remove special characters, keep alphanumeric characters and spaces
text = re.sub(r'[^a-z0-9\s]', ' ', text)
# Tokenize
tokens = text.split()
return tokens
def _encode(self, text):
"""Encode text"""
tokens = self._tokenize(text)[:self.max_len]
indices = [
self.vocab.get(token, self.vocab['<UNK>'])
for token in tokens
]
# Pad with zeros
if len(indices) < self.max_len:
indices += [self.vocab['<PAD>']] * (self.max_len - len(indices))
return indices
def __len__(self):
return len(self.texts)
def __getitem__(self, idx):
text = self.texts[idx]
label = self.labels[idx]
encoded = self._encode(text)
return torch.tensor(encoded, dtype=torch.long), torch.tensor(label, dtype=torch.long)
# Usage example
texts = [
"This is a great movie",
"I hated this film",
"Amazing performance",
"Terrible acting",
"Highly recommend",
]
labels = [1, 0, 1, 0, 1]
dataset = TextDataset(texts, labels, max_len=10)
print(f"Dataset size: {len(dataset)}")
print(f"Vocabulary size: {len(dataset.vocab)}")
# Get a sample
text, label = dataset[0]
print(f"Text encoding: {text[:5]}...")
print(f"Label: {label}")
3.2 Advanced Tokenizers
Example
def tokenize_with_spacy(text):
"""Tokenize using spacy"""
try:
import spacy
nlp = spacy.load("en_core_web_sm")
doc = nlp(text)
return [token.text for token in doc]
except ImportError:
print("spacy is not installed, using simple tokenization")
return text.lower().split()
# Use nltk for tokenization and stemming
def tokenize_with_nltk(text):
"""Tokenize using nltk"""
try:
import nltk
from nltk.tokenize import word_tokenize
from nltk.stem import PorterStemmer
# Tokenize
tokens = word_tokenize(text.lower())
# Optional: stemming
stemmer = PorterStemmer()
tokens = [stemmer.stem(token) for token in tokens if token.isalnum()]
return tokens
except ImportError:
return text.lower().split()
# Use SentencePiece for subword tokenization
class SentencePieceTokenizer:
"""
Use SentencePiece for subword tokenization
"""
def __init__(self, model_path=None):
self.model_path = model_path
self.sp = None
def train(self, texts, vocab_size=10000):
"""Train SentencePiece model"""
try:
import sentencepiece as spm
import tempfile
# Write to a temporary file
with tempfile.NamedTemporaryFile(mode='w', delete=False, suffix='.txt') as f:
for text in texts:
f.write(text + '\n')
temp_file = f.name
# Train
spm.SentencePieceTrainer.train(
input=temp_file,
model_prefix='spm_model',
vocab_size=vocab_size,
character_coverage=1.0,
model_type='unigram',
)
self.sp = spm.SentencePieceProcessor()
self.sp.load('spm_model.model')
except ImportError:
print("sentencepiece is not installed")
def encode(self, text):
if self.sp:
return self.sp.encode(text, out_type=str)
return text.split()
def decode(self, ids):
if self.sp:
return self.sp.decode(ids)
return ' '.join(ids)
3.3 Complete Data Loader
Example
from torch.utils.data import Dataset, DataLoader
from collections import Counter
class NLPDataset(Dataset):
"""
Complete NLP dataset class
Supports variable-length sequences, vocabulary building, and batch padding
"""
def __init__(self, texts, labels, min_freq=2, max_vocab=30000):
self.texts = texts
self.labels = labels
self.min_freq = min_freq
# Build vocabulary
self.word2idx, self.idx2word = self._build_vocab(texts, max_vocab)
self.pad_idx = self.word2idx['<PAD>']
self.unk_idx = self.word2idx['<UNK>']
def _tokenize(self, text):
text = text.lower()
text = re.sub(r'[^a-z0-9\s]', ' ', text)
return text.split()
def _build_vocab(self, texts, max_vocab):
"""Build vocabulary"""
counter = Counter()
for text in texts:
tokens = self._tokenize(text)
counter.update(tokens)
# Build vocabulary
word2idx = {'<PAD>': 0, '<UNK>': 1}
idx2word = {0: '<PAD>', 1: '<UNK>'}
for word, count in counter.most_common(max_vocab):
if count >= self.min_freq and word not in word2idx:
idx = len(word2idx)
word2idx[word] = idx
idx2word[idx] = word
return word2idx, idx2word
def __len__(self):
return len(self.texts)
def __getitem__(self, idx):
text = self.texts[idx]
tokens = self._tokenize(text)
indices = [self.word2idx.get(t, self.unk_idx) for t in tokens]
return {
'input': indices,
'label': self.labels[idx],
'length': len(indices)
}
def collate_fn(batch, pad_idx=0):
"""
Custom collate function: handle variable-length sequences
"""
# Sort by length in descending order
batch.sort(key=lambda x: x['length'], reverse=True)
texts = [item['input'] for item in batch]
labels = [item['label'] for item in batch]
lengths = [item['length'] for item in batch]
# Pad to the same length
max_len = max(lengths)
padded = [text + [pad_idx] * (max_len - len(text)) for text in texts]
return {
'input': torch.tensor(padded, dtype=torch.long),
'labels': torch.tensor(labels, dtype=torch.long),
'lengths': torch.tensor(lengths, dtype=torch.long)
}
# Use DataLoader
def create_dataloader(texts, labels, batch_size=32, shuffle=True):
"""Create data loader"""
dataset = NLPDataset(texts, labels)
pad_idx = dataset.pad_idx
dataloader = DataLoader(
dataset,
batch_size=batch_size,
shuffle=shuffle,
collate_fn=lambda b: collate_fn(b, pad_idx)
)
return dataloader, dataset
# Simulate data
texts = [
"I love this movie",
"This is a bad movie",
"Great film",
"Terrible acting",
"I recommend this",
"Not worth watching",
]
labels = [1, 0, 1, 0, 1, 0]
dataloader, dataset = create_dataloader(texts, labels, batch_size=2)
for batch in dataloader:
print("Input shape:", batch['input'].shape)
print("Labels:", batch['labels'])
print("Lengths:", batch['lengths'])
print("---")
4. Loading Pretrained Word Vectors
Using pretrained word vectors can improve model performance.
4.1 GloVe Word Vectors
Example
import numpy as np
def load_glove_vectors(filepath, word2idx, embedding_dim=300):
"""
Load GloVe pretrained word vectors
Parameters:
filepath: GloVe file path
word2idx: vocabulary dictionary
embedding_dim: word vector dimension
"""
print("Loading GloVe word vectors...")
# Initialize embedding matrix
num_words = len(word2idx)
embeddings = np.random.randn(num_words, embedding_dim).astype(np.float32)
embeddings[word2idx['<PAD>']] = np.zeros(embedding_dim) # Set PAD vector to zero
words_loaded = 0
with open(filepath, 'r', encoding='utf-8') as f:
for line in f:
values = line.strip().split()
word = values[0]
if word in word2idx:
idx = word2idx[word]
vector = np.asarray(values[1:], dtype=np.float32)
if len(vector) == embedding_dim:
embeddings[idx] = vector
words_loaded += 1
print(f"Loaded {words_loaded}/{num_words} word vectors")
return torch.from_numpy(embeddings)
# Simulated loading
def create_pretrained_embedding(word2idx, embedding_dim=300, pretrained_path=None):
"""
Create pretrained embedding layer
"""
if pretrained_path and False: # Set to True and provide a valid path
weights = load_glove_vectors(pretrained_path, word2idx, embedding_dim)
else:
# Use random initialization
num_words = len(word2idx)
weights = torch.randn(num_words, embedding_dim) * 0.1
embedding = torch.nn.Embedding.from_pretrained(weights, padding_idx=0)
return embedding
# Example
word2idx = {'<PAD>': 0, '<UNK>': 1, 'hello': 2, 'world': 3, 'test': 4}
embedding = create_pretrained_embedding(word2idx, embedding_dim=100)
print(f"Embedding layer shape: {embedding.weight.shape}")
4.2 Loading Word2Vec with gensim
Example
"""
Loading Word2Vec with gensim
"""
try:
from gensim.models import KeyedVectors
# Load the model
# model = KeyedVectors.load_word2vec_format(model_path, binary=True)
# Create the embedding matrix
embedding_dim = 300
num_words = len(word2idx)
embeddings = np.random.randn(num_words, embedding_dim).astype(np.float32)
# Fill in the word vectors
embeddings[word2idx['<PAD>']] = np.zeros(embedding_dim)
# Load the word vectors
# for word, idx in word2idx.items():
# if word in model:
# embeddings[idx] = model[word]
return torch.from_numpy(embeddings)
except ImportError:
print("gensim is not installed")
return None
4.3 Custom Pretrained Embeddings
Example
import torch.nn as nn
class PretrainedEmbedding(nn.Module):
"""
Embedding layer that supports loading pretrained word vectors
"""
def __init__(self, vocab_size, embed_dim, pretrained_weights=None,
padding_idx=0, freeze=False):
super().__init__()
self.embedding = nn.Embedding(
vocab_size,
embed_dim,
padding_idx=padding_idx
)
# Load pretrained weights
if pretrained_weights is not None:
self.embedding.weight.data.copy_(pretrained_weights)
# Whether to freeze
if freeze:
self.embedding.weight.requires_grad = False
def forward(self, x):
return self.embedding(x)
# Usage example
vocab_size = 10000
embed_dim = 300
pretrained = torch.randn(vocab_size, embed_dim) * 0.1
embedding_layer = PretrainedEmbedding(
vocab_size,
embed_dim,
pretrained_weights=pretrained,
padding_idx=0,
freeze=False # Set to True to freeze the embedding layer
)
# Test
x = torch.tensor([[1, 2, 3], [4, 5, 6]])
output = embedding_layer(x)
print(f"Input shape: {x.shape}")
print(f"Output shape: {output.shape}")
5. Text Data Augmentation
Data augmentation can improve model generalization ability.
5.1 Back-translation Augmentation
Example
def back_translation_augment(text, translator):
"""
Back-translation augmentation: translate text into another language and then translate it back
"""
# Translate to the target language
translated = translator.translate(text, dest='fr')
# Translate back to the source language
back_translated = translator.translate(translated.text, dest='en')
return back_translated.text
# Usage example
# from googletrans import Translator
# translator = Translator()
# augmented_text = back_translation_augment("I love this movie", translator)
5.2 Synonym Replacement
Example
def synonym_replacement(text, n=1, stop_words=None):
"""
Synonym replacement
"""
if stop_words is None:
stop_words = {'a', 'an', 'the', 'is', 'are', 'was', 'were'}
words = text.lower().split()
replaceable = [i for i, w in enumerate(words) if w not in stop_words]
if not replaceable:
return text
n = min(n, len(replaceable))
replace_idx = random.sample(replaceable, n)
# Simplified synonym mapping (should actually use WordNet)
synonym_map = {
'good': ['great', 'excellent', 'nice'],
'bad': ['poor', 'terrible', 'awful'],
'movie': ['film', 'cinema'],
'loved': ['adored', 'enjoyed'],
}
for idx in replace_idx:
word = words[idx]
if word in synonym_map:
words[idx] = random.choice(synonym_map[word])
return ' '.join(words)
# Usage example
text = "I loved this good movie"
augmented = synonym_replacement(text, n=2)
print(f"Original sentence: {text}")
print(f"Augmented: {augmented}")
5.3 Random Insertion and Deletion
Example
def random_insertion(text, n=1):
"""
Random insertion: randomly insert words into the text
"""
words = text.lower().split()
for _ in range(n):
if len(words) > 1:
idx = random.randint(0, len(words) - 1)
# Simplified inserted words
insert_words = ['really', 'very', 'quite', 'actually']
words.insert(idx, random.choice(insert_words))
return ' '.join(words)
def random_deletion(text, p=0.1):
"""
Random deletion: delete words with probability p
"""
words = text.lower().split()
if len(words) == 1:
return text
remaining = [w for w in words if random.random() > p]
if len(remaining) == 0:
return random.choice(words)
return ' '.join(remaining)
def random_swap(text, n=1):
"""
Random swap: randomly swap the positions of two words
"""
words = text.lower().split()
if len(words) < 2:
return text
for _ in range(n):
idx1, idx2 = random.sample(range(len(words)), 2)
words[idx1], words[idx2] = words[idx2], words[idx1]
return ' '.join(words)
# Usage example
text = "This is a great movie about love and happiness"
print("Original:", text)
print("Inserted:", random_insertion(text, n=2))
print("Deleted:", random_deletion(text, p=0.2))
print("Swapped:", random_swap(text, n=2))
6. Common NLP Task Examples
6.1 Complete Text Classification Pipeline
Example
import torch.nn as nn
from torch.utils.data import Dataset, DataLoader
# 1. Data preparation
texts = [
"I love this product, it is amazing!",
"Terrible quality, waste of money.",
"Great value for the price.",
"Not satisfied with the purchase.",
"Best purchase I have ever made!",
"Very disappointed with this item.",
]
labels = [1, 0, 1, 0, 1, 0] # 1: positive, 0: negative
# 2. Dataset
class TextClassificationDataset(Dataset):
def __init__(self, texts, labels, max_len=50):
self.texts = texts
self.labels = labels
self.max_len = max_len
self.vocab = self._build_vocab(texts)
def _build_vocab(self, texts):
vocab = {'<PAD>': 0, '<UNK>': 1}
for text in texts:
for word in text.lower().split():
if word not in vocab:
vocab[word] = len(vocab)
return vocab
def __len__(self):
return len(self.texts)
def __getitem__(self, idx):
text = self.texts[idx]
label = self.labels[idx]
# Simple encoding
tokens = text.lower().split()[:self.max_len]
indices = [self.vocab.get(t, 1) for t in tokens]
# Padding
if len(indices) < self.max_len:
indices += [0] * (self.max_len - len(indices))
return torch.tensor(indices, dtype=torch.long), torch.tensor(label, dtype=torch.long)
# 3. Model
class TextClassifier(nn.Module):
def __init__(self, vocab_size, embed_dim=128, hidden_dim=64, num_classes=2):
super().__init__()
self.embedding = nn.Embedding(vocab_size, embed_dim, padding_idx=0)
self.lstm = nn.LSTM(embed_dim, hidden_dim, batch_first=True)
self.fc = nn.Linear(hidden_dim, num_classes)
def forward(self, x):
embedded = self.embedding(x)
_, (hidden, _) = self.lstm(embedded)
output = self.fc(hidden.squeeze(0))
return output
# 4. Training
dataset = TextClassificationDataset(texts, labels)
dataloader = DataLoader(dataset, batch_size=2, shuffle=True)
model = TextClassifier(vocab_size=len(dataset.vocab))
criterion = nn.CrossEntropyLoss()
optimizer = torch.optim.Adam(model.parameters(), lr=0.001)
# Train for a few epochs
model.train()
for epoch in range(10):
total_loss = 0
for texts_batch, labels_batch in dataloader:
optimizer.zero_grad()
outputs = model(texts_batch)
loss = criterion(outputs, labels_batch)
loss.backward()
optimizer.step()
total_loss += loss.item()
print(f"Epoch {epoch+1}, Loss: {total_loss/len(dataloader):.4f}")
print("Training complete!")
6.2 Sequence Labeling (NER) Example
Example
import torch.nn as nn
from torch.utils.data import Dataset, DataLoader
# Sequence labeling dataset
class NERDataset(Dataset):
"""
Named entity recognition dataset
Format: [(word, tag), ...]
"""
def __init__(self, sentences_tags, word2idx, tag2idx, max_len=50):
self.sentences_tags = sentences_tags
self.word2idx = word2idx
self.tag2idx = tag2idx
self.max_len = max_len
def __len__(self):
return len(self.sentences_tags)
def __getitem__(self, idx):
sentence_tags = self.sentences_tags[idx]
words = [wt[0] for wt in sentence_tags]
tags = [wt[1] for wt in sentence_tags]
# Encoding
word_ids = [self.word2idx.get(w.lower(), 1) for w in words[:self.max_len]]
tag_ids = [self.tag2idx[t] for t in tags[:self.max_len]]
# Padding
if len(word_ids) < self.max_len:
word_ids += [0] * (self.max_len - len(word_ids))
tag_ids += [0] * (self.max_len - len(tag_ids))
return {
'words': torch.tensor(word_ids, dtype=torch.long),
'tags': torch.tensor(tag_ids, dtype=torch.long)
}
# NER model
class NERModel(nn.Module):
def __init__(self, vocab_size, tag_size, embed_dim=128, hidden_dim=256):
super().__init__()
self.embedding = nn.Embedding(vocab_size, embed_dim, padding_idx=0)
self.lstm = nn.LSTM(embed_dim, hidden_dim, batch_first=True, bidirectional=True)
self.fc = nn.Linear(hidden_dim * 2, tag_size)
def forward(self, x):
embedded = self.embedding(x)
lstm_out, _ = self.lstm(embedded)
logits = self.fc(lstm_out)
return logits
# Example data
data = [
[("John", "B-PER"), ("lives", "O"), ("in", "O"), ("New", "B-LOC"), ("York", "I-LOC"), (".", "O")],
[("Apple", "B-ORG"), ("is", "O"), ("a", "O"), ("company", "O"), (".", "O")],
]
# Build word vocabulary and tag table
word2idx = {'<PAD>': 0, '<UNK>': 1}
tag2idx = {'O': 0, 'B-PER': 1, 'I-PER': 2, 'B-LOC': 3, 'I-LOC': 4, 'B-ORG': 5, 'I-ORG': 6}
for sent in data:
for word, tag in sent:
if word.lower() not in word2idx:
word2idx[word.lower()] = len(word2idx)
print(f"Vocabulary size: {len(word2idx)}")
print(f"Number of tags: {len(tag2idx)}")
7. Recommended Alternatives
7.1 Other Text Processing Libraries
| Library name | Features | Applicable scenarios |
|---|---|---|
| HuggingFace Datasets | Data loading and processing | General NLP tasks |
| spacy | Advanced NLP tools | Tokenization, NER, dependency parsing |
| NLTK | Classic NLP tools | Teaching, prototyping |
| textblob | Simple and easy to use | Fast text processing |
7.2 Using HuggingFace Datasets
Example
# pip install datasets
from datasets import load_dataset
# Load the dataset
# dataset = load_dataset("imdb")
# dataset = load_dataset("squad")
# Data preprocessing
# def preprocess_function(examples):
# return tokenizer(examples['text'], truncation=True, padding='max_length')
# tokenized_dataset = dataset.map(preprocess_function, batched=True)
print("HuggingFace Datasets usage example:")
print("from datasets import load_dataset")
print("dataset = load_dataset('imdb')")
8. API Quick Reference
8.1 Common Text Processing Operations
| Operation | Method |
|---|---|
| Tokenization | text.split(), spacy, nltk, jieba |
| Vocabulary building | Counter, vocab.build_vocab() |
| Encoding | vocab.encode(), tokenizer.encode() |
| Decoding | vocab.decode(), tokenizer.decode() |
| Padding | pad_sequence, collate_fn |
| Loading word vectors | Gensim, GloVe direct loading |
8.2 Data Augmentation Methods
| Method | Description |
|---|---|
| Synonym replacement | Randomly replace words with synonyms |
| Random insertion | Randomly insert words |
| Random deletion | Randomly delete words |
| Random swap | Randomly swap word positions |
| Back-translation | Translate and then translate back |
8.3 Recommended Data Processing Pipeline
1. 数据清洗 - 去除 HTML 标签、特殊字符 - 统一编码、大小写 2. 分词 - 英文: spaCy / NLTK - 中文: jieba / pkuseg 3. 词表构建 - 过滤低频词 - 设定最大词表大小 4. 序列化 - 编码为索引 - 统一长度(填充/截断) 5. 数据增强(可选) - 同义词替换、回译 6. 批量加载 - DataLoader + 自定义 collate_fnOther extensions