## ----include = FALSE----------------------------------------------------------
knitr::opts_chunk$set(
  collapse = TRUE,
  message = FALSE,
  warning = FALSE,
  fig.width = 8,
  fig.height = 5,
  comment = "#>"
)

## ----setup--------------------------------------------------------------------
library(tidyEmoji)
library(dplyr)

# The charts below need three packages that tidyEmoji only Suggests. A vignette
# has to build without its optional dependencies, so the plotting chunks are
# gated on this flag rather than assuming the packages are there.
has_plot_pkgs <- all(vapply(
  c("ggplot2", "forcats", "stringr"),
  requireNamespace, logical(1), quietly = TRUE
))
if (has_plot_pkgs) library(ggplot2)

## -----------------------------------------------------------------------------
# read.csv() rather than readr::read_csv(): readr is a Suggests package, and
# reading one CSV does not need it. system.file() rather than a bare path, so
# this chunk works from the installed introduction.R as well as at build time
# -- the corpus lives in inst/extdata and is installed with the package.
ata_tweets <- tibble::as_tibble(
  utils::read.csv(
    system.file("extdata", "ata_tweets.csv", package = "tidyEmoji"),
    encoding = "UTF-8", stringsAsFactors = FALSE
  )
)
ata_tweets

## -----------------------------------------------------------------------------
summary_tbl <- ata_tweets %>%
  emoji_summary(full_text)

summary_tbl

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_filter(full_text)

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_extract_nest(full_text) %>%
  select(full_text, .emoji_unicode)

## -----------------------------------------------------------------------------
emoji_per_tweet <- ata_tweets %>%
  emoji_extract_unnest(full_text)

emoji_per_tweet

## ----eval = has_plot_pkgs, fig.alt = "Bar chart of the number of emoji per emoji-bearing entry. About two-thirds of entries contain a single emoji, with a long, thin tail of more emoji-heavy entries."----
emoji_per_tweet %>%
  group_by(.row_number) %>%
  summarise(n_emoji = sum(.emoji_count)) %>%
  ggplot(aes(n_emoji)) +
  geom_bar() +
  scale_x_continuous(breaks = seq(1, 15)) +
  labs(x = "Number of emoji in the entry",
       y = "Number of entries",
       title = "Most emoji-bearing entries contain a single emoji")

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_tokens(full_text)

## -----------------------------------------------------------------------------
demo <- data.frame(
  text = c("our family \U0001F468‍\U0001F469‍\U0001F467‍\U0001F466",
           "great work \U0001F44D\U0001F3FD")
)

demo %>%
  emoji_extract_unnest(text)

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_frequency(full_text)

## -----------------------------------------------------------------------------
top_20_emojis <- ata_tweets %>%
  top_n_emojis(full_text)

top_20_emojis

## ----eval = has_plot_pkgs, fig.alt = "Horizontal bar chart of the 20 most frequent emoji in the corpus, coloured by Unicode category."----
top_20_emojis %>%
  mutate(emoji_name = stringr::str_replace_all(emoji_name, "_", " "),
         emoji_name = forcats::fct_reorder(emoji_name, n)) %>%
  ggplot(aes(n, emoji_name, fill = emoji_category)) +
  geom_col() +
  labs(x = "Count",
       y = NULL,
       fill = "Category",
       title = "The 20 most frequent emoji")

## ----eval = has_plot_pkgs, fig.alt = "Horizontal bar chart of the 10 most frequent emoji in the corpus, coloured by Unicode category."----
ata_tweets %>%
  top_n_emojis(full_text, n = 10) %>%
  mutate(emoji_name = stringr::str_replace_all(emoji_name, "_", " "),
         emoji_name = forcats::fct_reorder(emoji_name, n)) %>%
  ggplot(aes(n, emoji_name, fill = emoji_category)) +
  geom_col() +
  labs(x = "Count", y = NULL, fill = "Category",
       title = "The 10 most frequent emoji")

## -----------------------------------------------------------------------------
ata_emoji_category <- ata_tweets %>%
  emoji_categorize(full_text) %>%
  select(.emoji_category)

ata_emoji_category

## ----eval = has_plot_pkgs, fig.alt = "Horizontal bar chart of the most common emoji category combinations that appear in more than 20 entries."----
ata_emoji_category %>%
  count(.emoji_category, sort = TRUE) %>%
  filter(n > 20) %>%
  mutate(.emoji_category = forcats::fct_reorder(.emoji_category, n)) %>%
  ggplot(aes(n, .emoji_category)) +
  geom_col() +
  labs(x = "Number of entries", y = NULL,
       title = "Most common emoji category combinations")

## ----eval = has_plot_pkgs, fig.alt = "Horizontal bar chart of how often each individual Unicode emoji category is used, dominated by Smileys & Emotion followed by People & Body."----
ata_emoji_category %>%
  tidyr::separate_longer_delim(.emoji_category, delim = "|") %>%
  count(.emoji_category, sort = TRUE) %>%
  mutate(.emoji_category = forcats::fct_reorder(.emoji_category, n)) %>%
  ggplot(aes(n, .emoji_category)) +
  geom_col() +
  labs(x = "Number of entries", y = NULL,
       title = "Emoji category usage")

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_type(full_text) %>%
  count(.emoji_type, sort = TRUE) %>%
  head(5)

ata_tweets %>%
  emoji_faceness(full_text) %>%
  summarise(mean_faceness = mean(.emoji_faceness, na.rm = TRUE),
            all_faces = sum(.emoji_faceness == 1, na.rm = TRUE))

## -----------------------------------------------------------------------------
ata_sentiment <- ata_tweets %>%
  emoji_sentiment(full_text)

ata_sentiment %>%
  select(.emoji_n, .emoji_sentiment)

## ----eval = has_plot_pkgs, fig.alt = "Histogram of the mean emoji sentiment per entry, which is concentrated on the positive side of the scale."----
ata_sentiment %>%
  filter(!is.na(.emoji_sentiment)) %>%
  ggplot(aes(.emoji_sentiment)) +
  geom_histogram(binwidth = 0.1) +
  labs(x = "Mean emoji sentiment",
       y = "Number of entries",
       title = "Emoji sentiment skews positive")

## ----eval = has_plot_pkgs, fig.alt = "Horizontal bar chart of the average emoji sentiment within each Unicode category."----
ata_tweets %>%
  emoji_tokens(full_text) %>%
  group_by(.emoji_category) %>%
  summarise(mean_sentiment = mean(.emoji_sentiment, na.rm = TRUE),
            n_scored = sum(!is.na(.emoji_sentiment))) %>%
  filter(n_scored > 0) %>%
  mutate(.emoji_category = forcats::fct_reorder(.emoji_category, mean_sentiment)) %>%
  ggplot(aes(mean_sentiment, .emoji_category)) +
  geom_col() +
  labs(x = "Mean sentiment", y = NULL,
       title = "Average emoji sentiment by category")

## -----------------------------------------------------------------------------
emoji_sentiment_lexicon %>%
  filter(occurrences >= 500) %>%
  slice_max(sentiment_score, n = 8) %>%
  select(emoji, unicode_name, occurrences, sentiment_score)

emoji_sentiment_lexicon %>%
  filter(occurrences >= 500) %>%
  slice_min(sentiment_score, n = 8) %>%
  select(emoji, unicode_name, occurrences, sentiment_score)

## -----------------------------------------------------------------------------
emoji_ambiguity() %>%
  filter(n_annotations > 500) %>%
  head(5)

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_flag_ambiguous(full_text, top_n = 5)

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_risk(full_text) %>%
  filter(.emoji_n > 1) %>%
  select(.emoji_n, .emoji_n_scored, .emoji_ambiguity_mean,
         .emoji_n_ambiguous) %>%
  head(5)

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_sentiment(full_text, se = TRUE) %>%
  filter(!is.na(.emoji_sentiment)) %>%
  select(.emoji_n_scored, .emoji_sentiment, .emoji_sentiment_se) %>%
  head(5)

## -----------------------------------------------------------------------------
ata_emotion <- ata_tweets %>%
  emoji_emotion(full_text)

ata_emotion %>%
  select(.emoji_joy, .emoji_trust, .emoji_anger, .emoji_n)

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_emotion_label(full_text) %>%
  count(.emoji_emotion, sort = TRUE)

## -----------------------------------------------------------------------------
emoji_lexicons()

## -----------------------------------------------------------------------------
my_lexicon <- data.frame(
  emoji = c("\U0001f600", "\U0001f621", "\U0001f637"),
  score = c(1, -1, -0.5)
)

data.frame(text = c("great \U0001f600", "bad \U0001f621\U0001f637", "none")) %>%
  emoji_score(text, lexicon = my_lexicon)

## -----------------------------------------------------------------------------
register_emoji_lexicon("mine", my_lexicon)
emoji_lexicons() %>% filter(name == "mine")

## -----------------------------------------------------------------------------
emoji_edges <- ata_tweets %>%
  emoji_pairs(full_text)

emoji_edges

## ----eval = has_plot_pkgs, fig.alt = "Horizontal bar chart of the most frequent emoji pairs, labelled by the two glyphs of each pair."----
emoji_edges %>%
  slice_max(n, n = 10) %>%
  mutate(pair = paste(item1, item2),
         pair = forcats::fct_reorder(pair, n)) %>%
  ggplot(aes(n, pair)) +
  geom_col() +
  labs(x = "Number of entries containing both", y = NULL,
       title = "Emoji that appear together")

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_ngrams(full_text) %>%
  count(.emoji_ngram, sort = TRUE)

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_context(full_text, window = 4) %>%
  select(.row_number, .emoji, .emoji_context) %>%
  head(5)

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_collocations(full_text, window = 4, min_n = 5) %>%
  head(10)

## ----eval = has_plot_pkgs, fig.alt = "Histogram of the mean relative position of emoji within each entry, showing emoji concentrated towards the end of the text."----
ata_tweets %>%
  emoji_position(full_text) %>%
  filter(!is.na(.emoji_rel_position)) %>%
  ggplot(aes(.emoji_rel_position)) +
  geom_histogram(binwidth = 0.05) +
  labs(x = "Mean relative position of the entry's emoji",
       y = "Number of entries",
       title = "Emoji cluster at the end of a message")

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_ratio(full_text) %>%
  summarise(
    n_emoji_only = sum(.emoji_only, na.rm = TRUE),
    mean_ratio   = mean(.emoji_ratio[.emoji_ratio > 0], na.rm = TRUE)
  )

## -----------------------------------------------------------------------------
dated <- ata_tweets %>%
  mutate(posted_at = as.Date("2021-01-01") + (seq_len(n()) - 1) %% 540)

dated %>%
  emoji_trend(full_text, posted_at, by = "quarter", top_n = 3)

## -----------------------------------------------------------------------------
dated %>%
  emoji_turnover(full_text, posted_at, by = "quarter")

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_version_profile(full_text) %>%
  head(8)

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_dfm(full_text, weighting = "tfidf") %>%
  select(1:6)

## -----------------------------------------------------------------------------
positive <- c("love", "great", "best", "happy", "good", "thanks", "beautiful")
negative <- c("hate", "worst", "bad", "sad", "awful", "sick", "tired")

scored <- ata_tweets %>%
  mutate(text_score = vapply(
    strsplit(tolower(full_text), "[^a-z]+"),
    function(w) as.numeric(sum(w %in% positive) - sum(w %in% negative)),
    numeric(1)
  ))

## -----------------------------------------------------------------------------
incong <- scored %>%
  emoji_incongruity(full_text, text_score, scale = "rank")

incong %>%
  filter(!is.na(.emoji_incongruity)) %>%
  count(.emoji_polarity_flip)

incong %>%
  filter(.emoji_polarity_flip) %>%
  select(full_text, .emoji_sentiment, text_score) %>%
  head(3)

## -----------------------------------------------------------------------------
scored %>%
  emoji_incongruity_profile(full_text, text_score, scale = "rank", min_n = 10)

## -----------------------------------------------------------------------------
demo <- data.frame(text = "great \U0001f600 love \u2764\ufe0f")
demo %>% emoji_to_text(text, format = "name")
demo %>% emoji_to_text(text, format = "shortcode")
demo %>%
  emoji_to_text(text, format = "shortcode") %>%
  text_to_emoji(text)

## -----------------------------------------------------------------------------
as_emoji_name(c("\U0001f600", "\u2764\ufe0f"))
as_emoji_shortcode(c("\U0001f600", "\u2764\ufe0f"))
as_emoji(c("grinning", "heart"))

## -----------------------------------------------------------------------------
emoji_search("happy")
emoji_search("celebration")

## -----------------------------------------------------------------------------
ata_tweets %>%
  emoji_token_cost(full_text) %>%
  filter(.emoji_n > 0) %>%
  summarise(emoji = sum(.emoji_n),
            bytes = sum(.emoji_bytes),
            codepoints = sum(.emoji_codepoints),
            est_tokens = sum(.emoji_token_estimate))

## -----------------------------------------------------------------------------
demo_llm <- data.frame(text = "ship it \U0001f680 today")
for (p in c("keep", "strip", "name", "shortcode", "placeholder")) {
  cat(format(p, width = 12), emoji_sanitize(demo_llm, text, policy = p)$text,
      "\n")
}

## -----------------------------------------------------------------------------
emoji_provenance() %>% glimpse()

