Rで学ぶソーシャルメディアデータ分析
Vivek Vijayaraghavan
Data Science Coach
-filter でオリジナルツイートを抽出-filter:retweets でリツイート除外-filter:quote で引用を除外-filter:replies で返信を除外# 「digital marketing」のツイートを100件抽出
tweets_all <- search_tweets("digital marketing", n = 100)
reply_to_screen_name、is_quote、is_retweet の値を集計# 返信数を確認
library(plyr)
count(tweets_all$reply_to_screen_name)
x freq
<fct> <int>
blairaasmith 2
javiergosende 1
juanburgos 1
WhutTheHale 2
NA 94
# 引用ツイート数を確認
count(tweets_all$is_quote)
x freq
<lgl> <int>
FALSE 98
TRUE 2
# リツイート数を確認
count(tweets_all$is_retweet)
x freq
<lgl> <int>
FALSE 61
TRUE 39
-filter で抽出# '-filter' を適用
tweets_org <- search_tweets("digital marketing
-filter:retweets
-filter:quote
-filter:replies",
n = 100)
# 返信数を確認
library(plyr)
count(tweets_org$reply_to_screen_name)
x freq
<lgl> <int>
NA 100
# 引用ツイート数を確認
library(plyr)
count(tweets_org$is_quote)
x freq
<lgl> <int>
FALSE 100
# リツイート数を確認
library(plyr)
count(tweets_org$is_retweet)
x freq
<lgl> <int>
FALSE 100
lang は言語でツイートをフィルタ
# スペイン語のツイートを抽出
tweets_lang <- search_tweets("brand marketing", lang = "es")
View(tweets_lang)

head(tweets_lang$lang)
[1] "es" "es" "es" "es" "es" "es"
min_faves: 最小いいね数でフィルタmin_retweets: 最小リツイート数でフィルタ AND を使用# いいね・リツイート各100以上を抽出
tweets_pop <- search_tweets("bitcoin min_faves:100 AND
min_retweets:100")
# リツイート・いいね数を確認するデータフレーム
counts <- tweets_pop[c("retweet_count", "favorite_count")]
head(counts)
retweet_count favorite_count
<int> <int>
1 162 833
2 141 894
3 164 1128
4 395 1346
5 475 2271
6 270 1654
# ツイート本文を表示
head(tweets_pop$text)
text
<chr>
1 As we continue to build the Bakkt Bitcoin Futures contract, we reached a
2 BREAKING: The United States is considering entering into a "currency pact"
3 REMINDER: The Bitcoin ETF will eventually get approved.\n\nNot a question
4 [New Post] Bitcoin is becoming much more important in Hong Kong and India.
5 Reports are surfacing that some Hong Kong ATMs have run out of cash as
6 Bitcoin is the most transparent currency ever created.
Rで学ぶソーシャルメディアデータ分析