Xử lý Ngôn ngữ Nói bằng Python
Daniel Bourke
Machine Learning Engineer/YouTube creator
# Kiểm tra thư mục audio sau mua
import os
post_purchase_audio = os.listdir("post_purchase")
print(post_purchase_audio[:5])
['post-purchase-audio-0.mp3',
'post-purchase-audio-1.mp3',
'post-purchase-audio-2.mp3',
'post-purchase-audio-3.mp3',
'post-purchase-audio-4.mp3']
# Lặp qua các tệp mp3
for file in post_purchase_audio:
print(f"Converting {file} to .wav...")
# Dùng hàm đã tạo để chuyển sang .wav
convert_to_wav(file)
Converting post-purchase-audio-0.mp3 to .wav...
Converting post-purchase-audio-1.mp3 to .wav...
Converting post-purchase-audio-2.mp3 to .wav...
Converting post-purchase-audio-3.mp3 to .wav...
Converting post-purchase-audio-4.mp3 to .wav...
# Phiên âm văn bản từ tệp wav def create_text_list(folder):text_list = []# Lặp qua thư mục for file in folder:# Kiểm tra phần mở rộng .wav if file.endswith(".wav"):# Phiên âm audio text = transcribe_audio(file)# Thêm văn bản đã phiên âm vào danh sách text_list.append(text)return text_list
# Chuyển audio sau mua thành văn bản post_purchase_text = create_text_list(post_purchase_audio)print(post_purchase_text[:5])
['hey man I just water product from you guys and I think is amazing but I leave a little help setting it up',
'these clothes I just bought from you guys too small is there anyway I can change the size',
"I recently got these pair of shoes but they're too big can I change the size",
"I bought a pair of pants from you guys but they're way too small",
"I bought a pair of pants and they're the wrong colour is there any chance I can change that"]
import pandas as pd# Tạo dataframe sau mua post_purchase_df = pd.DataFrame({"label": "post_purchase", "text": post_purchase_text})# Tạo dataframe trước mua pre_purchase_df = pd.DataFrame({"label": "pre_purchase", "text": pre_purchase_text})
# Gộp trước mua và sau mua
df = pd.concat([post_purchase_df, pre_purchase_df])
# Xem dataframe đã gộp
df.head()
label text
0 post_purchase yeah hello someone this morning delivered a pa...
1 post_purchase my shipment arrived yesterday but it's not the...
2 post_purchase hey my name is Daniel I received my shipment y...
3 post_purchase hey mate how are you doing I'm just calling in...
4 pre_purchase hey I was wondering if you know where my new p...
# Import text classification packages
import numpy as np
from sklearn.pipeline import Pipeline
from sklearn.naive_bayes import MultinomialNB
from sklearn.feature_extraction.text import CountVectorizer, TfidfTransformer
from sklearn.model_selection import train_test_split
# Chia dữ liệu thành train và test
X_train, X_test, y_train, y_test = train_test_split(
X=df["text"],
y=df["label"],
test_size=0.3)
# Tạo pipeline phân loại văn bản
text_classifier = Pipeline([
("vectorizer", CountVectorizer()),
("tfidf", TfidfTransformer()),
("classifier", MultinomialNB())
])
# Huấn luyện pipeline trên dữ liệu train
text_classifier.fit(X_train, y_train)
# Dự đoán và so sánh với nhãn test predictions = text_classifier.predict(X_test)accuracy = 100 * np.mean(predictions == y_test.label) print(f"The model is {accuracy:.2f}% accurate.")
Mô hình đạt độ chính xác 97.87%.
Xử lý Ngôn ngữ Nói bằng Python