Last active
March 10, 2021 09:47
-
-
Save rich-iannone/6f1a7652d2d57567937a to your computer and use it in GitHub Desktop.
Plotting SF salaries using the DiagrammeR R package. Dataset is available from Kaggle Datasets.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| library(DiagrammeR) | |
| library(DiagrammeRsvg) | |
| library(magrittr) | |
| # The CSV file is located inside a zip file at: | |
| # https://www.kaggle.com/kaggle/sf-salaries/downloads/ | |
| # sf-salaries-release-2015-12-21-03-21-32.zip | |
| salaries <- read.csv("Salaries.csv", | |
| stringsAsFactors = FALSE) | |
| # Use only salary data from 2014 | |
| salaries <- subset(salaries, Year == 2014) | |
| # Select the `Pay` columns from the data frame | |
| salaries_pay <- salaries[,4:6] | |
| # Ensure that the `Pay` columns are numeric | |
| salaries_pay$BasePay <- | |
| as.numeric(salaries_pay$BasePay) | |
| salaries_pay$OvertimePay <- | |
| as.numeric(salaries_pay$OvertimePay) | |
| salaries_pay$OtherPay <- | |
| as.numeric(salaries_pay$OtherPay) | |
| # Select only complete cases | |
| salaries_pay <- | |
| salaries_pay[which(complete.cases(salaries_pay)),] | |
| # Take a sample of 5000 values | |
| set.seed <- 20 | |
| salaries_pay_sample <- | |
| salaries_pay[sample(1:nrow(salaries_pay), 5000, FALSE),] | |
| # Create a series of (x, y) data points as an NDF, | |
| # make a scatterplot, and then save the file as a PDF | |
| create_xy_pts( | |
| series_label = "pay", | |
| x = as.numeric(salaries_pay_sample$BasePay), | |
| y = as.numeric(salaries_pay_sample$OvertimePay) + | |
| as.numeric(salaries_pay_sample$OtherPay), | |
| line_width = 1.0, | |
| width = 0.05, | |
| height = 0.05, | |
| shape = "circle", | |
| fill_color = "#00c5cd50", | |
| line_color = "#00868b50") %>% | |
| create_xy_graph( | |
| x_name = "Base Pay (USD)", | |
| y_name = "Overtime + Other Pay (USD)", | |
| heading = c("#San Francisco Salaries in 2014*", | |
| "#####Published February 11, 2016"), | |
| footer = "####*Source: Kaggle Datasets (https://www.kaggle.com/kaggle/sf-salaries).", | |
| xy_value_labels = c("USD:K", "USD:K"), | |
| include_xy_minima = c(FALSE, FALSE), | |
| include_legend = FALSE) %>% | |
| export_graph("salaries_sf_scatterplot.pdf") |
Author
Here is a piped version
library(DiagrammeR)
library(DiagrammeRsvg)
library(magrittr)
library(dplyr)
library(readr)
# The CSV file is located inside a zip file at:
# https://www.kaggle.com/kaggle/sf-salaries/downloads/
# sf-salaries-release-2015-12-21-03-21-32.zip
all_salaries <-
'https://gist.githubusercontent.com/abresler/3ecbe3898017eb13e451/raw/b66d95d81c0e5f95297bcc8346469b44d620a0f4/sf_salaries.csv' %>%
read_csv()
# Use only salary data from 2014
salaries <-
all_salaries %>%
dplyr::filter(Year == 2014)
# Select the `Pay` columns from the data frame
salaries_pay <-
salaries %>%
dplyr::select(BasePay:OtherPay) %>%
mutate_each(funs(as.numeric(.)), contains("Pay"))
salaries_pay %<>%
dplyr::filter(complete.cases(salaries_pay))
# Select only complete cases
# Take a sample of 5000 values
set.seed <-
20
salaries_pay_sample <-
salaries_pay %>%
sample_n(5000)
# Create a series of (x, y) data points as an NDF,
# make a scatterplot, and then save the file as a PDF
create_xy_pts(
series_label = "pay",
x =
salaries_pay_sample$BasePay,
y =
salaries_pay_sample$OvertimePay +
salaries_pay_sample$OtherPay,
line_width = 1.0,
width = 0.05,
height = 0.05,
shape = "circle",
fill_color = "#00c5cd50",
line_color = "#00868b50") %>%
create_xy_graph(
x_name = "Base Pay (USD)",
y_name = "Overtime + Other Pay (USD)",
heading = c("#San Francisco Salaries in 2014*",
"#####Published February 11, 2016"),
footer = "####*Source: Kaggle Datasets (https://www.kaggle.com/kaggle/sf-salaries).",
xy_value_labels = c("USD:K", "USD:K"),
include_xy_minima = c(FALSE, FALSE),
include_legend = FALSE) %>%
export_graph("salaries_sf_scatterplot.pdf")
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment
The graph: