galaaz 2.1.7 → 2.1.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +9 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot.Rmd +63 -58
- data/blogs/galaaz_ggplot/galaaz_ggplot.log +59 -68
- data/blogs/galaaz_ggplot/galaaz_ggplot.md +91 -84
- data/blogs/galaaz_ggplot/galaaz_ggplot.tex +125 -94
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-html/midwest_rb.png +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-html/scatter_plot_rb.png +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-markdown_github/midwest_rb.png +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-markdown_github/scatter_plot_rb.png +0 -0
- data/blogs/gknit/gknit.Rmd +33 -28
- data/blogs/gknit/gknit.md +47 -42
- data/blogs/gknit/gknit.tex +1368 -0
- data/blogs/gknit/gknit_files/figure-html/bubble-1.png +0 -0
- data/blogs/gknit/gknit_files/figure-html/diverging_bar.png +0 -0
- data/blogs/gknit/gknit_files/figure-latex/bubble-1.png +0 -0
- data/blogs/gknit/gknit_files/gknit_files/figure-latex/bubble-1.png +0 -0
- data/blogs/manual/manual.Rmd +129 -60
- data/blogs/manual/manual.log +289 -545
- data/blogs/manual/manual.md +551 -467
- data/blogs/manual/manual.tex +1059 -485
- data/blogs/manual/manual_files/figure-html/bubble-1.png +0 -0
- data/blogs/manual/manual_files/figure-latex/bubble-1.png +0 -0
- data/blogs/manual/manual_files/figure-markdown_github/bubble-1.png +0 -0
- data/blogs/manual/manual_files/figure-markdown_github/diverging_bar.png +0 -0
- data/blogs/manual/manual_files/manual_files/figure-latex/bubble-1.png +0 -0
- data/blogs/nse_dplyr/nse_dplyr.Rmd +28 -7
- data/blogs/nse_dplyr/nse_dplyr.log +49 -153
- data/blogs/nse_dplyr/nse_dplyr.md +676 -705
- data/blogs/nse_dplyr/nse_dplyr.tex +1589 -0
- data/blogs/oh_my/oh_my.Rmd +193 -55
- data/blogs/oh_my/oh_my.log +265 -95
- data/blogs/oh_my/oh_my.md +236 -95
- data/blogs/oh_my/oh_my.tex +1976 -68
- data/blogs/ruby_plot/ruby_plot.Rmd +42 -34
- data/blogs/ruby_plot/ruby_plot.log +101 -99
- data/blogs/ruby_plot/ruby_plot.md +52 -46
- data/blogs/ruby_plot/ruby_plot.tex +134 -102
- data/blogs/ruby_plot/ruby_plot_files/figure-html/dose_len.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_delivery.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_dose.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color2.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_decorations.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_points.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/final_box_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/final_violin_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/violin_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/dose_len.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facet_by_delivery.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facet_by_dose.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_by_delivery_color.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_by_delivery_color2.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_decorations.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_points.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/final_box_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/final_violin_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/violin_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/dose_len.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/facet_by_delivery.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/facet_by_dose.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/facets_by_delivery_color.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/facets_by_delivery_color2.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/facets_with_decorations.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/facets_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/facets_with_points.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/final_box_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/final_violin_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/ruby_plot_files/figure-latex/violin_with_jitter.png +0 -0
- data/lib/galaaz/cli.rb +51 -10
- data/script/omarchy/README.md +1 -1
- data/script/omarchy/galaaz-guide.sh +1 -1
- data/sty/galaaz.sty +22 -0
- data/version.rb +1 -1
- metadata +19 -1
|
@@ -0,0 +1,1589 @@
|
|
|
1
|
+
% Options for packages loaded elsewhere
|
|
2
|
+
\PassOptionsToPackage{unicode}{hyperref}
|
|
3
|
+
\PassOptionsToPackage{hyphens}{url}
|
|
4
|
+
\documentclass[
|
|
5
|
+
11pt,
|
|
6
|
+
]{article}
|
|
7
|
+
\usepackage{xcolor}
|
|
8
|
+
\usepackage[margin=1in]{geometry}
|
|
9
|
+
\usepackage{amsmath,amssymb}
|
|
10
|
+
\setcounter{secnumdepth}{5}
|
|
11
|
+
\usepackage{iftex}
|
|
12
|
+
\ifPDFTeX
|
|
13
|
+
\usepackage[T1]{fontenc}
|
|
14
|
+
\usepackage[utf8]{inputenc}
|
|
15
|
+
\usepackage{textcomp} % provide euro and other symbols
|
|
16
|
+
\else % if luatex or xetex
|
|
17
|
+
\usepackage{unicode-math} % this also loads fontspec
|
|
18
|
+
\defaultfontfeatures{Scale=MatchLowercase}
|
|
19
|
+
\defaultfontfeatures[\rmfamily]{Ligatures=TeX,Scale=1}
|
|
20
|
+
\fi
|
|
21
|
+
\usepackage{lmodern}
|
|
22
|
+
\ifPDFTeX\else
|
|
23
|
+
% xetex/luatex font selection
|
|
24
|
+
\fi
|
|
25
|
+
% Use upquote if available, for straight quotes in verbatim environments
|
|
26
|
+
\IfFileExists{upquote.sty}{\usepackage{upquote}}{}
|
|
27
|
+
\IfFileExists{microtype.sty}{% use microtype if available
|
|
28
|
+
\usepackage[]{microtype}
|
|
29
|
+
\UseMicrotypeSet[protrusion]{basicmath} % disable protrusion for tt fonts
|
|
30
|
+
}{}
|
|
31
|
+
\makeatletter
|
|
32
|
+
\@ifundefined{KOMAClassName}{% if non-KOMA class
|
|
33
|
+
\IfFileExists{parskip.sty}{%
|
|
34
|
+
\usepackage{parskip}
|
|
35
|
+
}{% else
|
|
36
|
+
\setlength{\parindent}{0pt}
|
|
37
|
+
\setlength{\parskip}{6pt plus 2pt minus 1pt}}
|
|
38
|
+
}{% if KOMA class
|
|
39
|
+
\KOMAoptions{parskip=half}}
|
|
40
|
+
\makeatother
|
|
41
|
+
\usepackage{color}
|
|
42
|
+
\usepackage{fancyvrb}
|
|
43
|
+
\newcommand{\VerbBar}{|}
|
|
44
|
+
\newcommand{\VERB}{\Verb[commandchars=\\\{\}]}
|
|
45
|
+
\DefineVerbatimEnvironment{Highlighting}{Verbatim}{commandchars=\\\{\}}
|
|
46
|
+
% Add ',fontsize=\small' for more characters per line
|
|
47
|
+
\usepackage{framed}
|
|
48
|
+
\definecolor{shadecolor}{RGB}{248,248,248}
|
|
49
|
+
\newenvironment{Shaded}{\begin{snugshade}}{\end{snugshade}}
|
|
50
|
+
\newcommand{\AlertTok}[1]{\textcolor[rgb]{0.94,0.16,0.16}{#1}}
|
|
51
|
+
\newcommand{\AnnotationTok}[1]{\textcolor[rgb]{0.56,0.35,0.01}{\textbf{\textit{#1}}}}
|
|
52
|
+
\newcommand{\AttributeTok}[1]{\textcolor[rgb]{0.13,0.29,0.53}{#1}}
|
|
53
|
+
\newcommand{\BaseNTok}[1]{\textcolor[rgb]{0.00,0.00,0.81}{#1}}
|
|
54
|
+
\newcommand{\BuiltInTok}[1]{#1}
|
|
55
|
+
\newcommand{\CharTok}[1]{\textcolor[rgb]{0.31,0.60,0.02}{#1}}
|
|
56
|
+
\newcommand{\CommentTok}[1]{\textcolor[rgb]{0.56,0.35,0.01}{\textit{#1}}}
|
|
57
|
+
\newcommand{\CommentVarTok}[1]{\textcolor[rgb]{0.56,0.35,0.01}{\textbf{\textit{#1}}}}
|
|
58
|
+
\newcommand{\ConstantTok}[1]{\textcolor[rgb]{0.56,0.35,0.01}{#1}}
|
|
59
|
+
\newcommand{\ControlFlowTok}[1]{\textcolor[rgb]{0.13,0.29,0.53}{\textbf{#1}}}
|
|
60
|
+
\newcommand{\DataTypeTok}[1]{\textcolor[rgb]{0.13,0.29,0.53}{#1}}
|
|
61
|
+
\newcommand{\DecValTok}[1]{\textcolor[rgb]{0.00,0.00,0.81}{#1}}
|
|
62
|
+
\newcommand{\DocumentationTok}[1]{\textcolor[rgb]{0.56,0.35,0.01}{\textbf{\textit{#1}}}}
|
|
63
|
+
\newcommand{\ErrorTok}[1]{\textcolor[rgb]{0.64,0.00,0.00}{\textbf{#1}}}
|
|
64
|
+
\newcommand{\ExtensionTok}[1]{#1}
|
|
65
|
+
\newcommand{\FloatTok}[1]{\textcolor[rgb]{0.00,0.00,0.81}{#1}}
|
|
66
|
+
\newcommand{\FunctionTok}[1]{\textcolor[rgb]{0.13,0.29,0.53}{\textbf{#1}}}
|
|
67
|
+
\newcommand{\ImportTok}[1]{#1}
|
|
68
|
+
\newcommand{\InformationTok}[1]{\textcolor[rgb]{0.56,0.35,0.01}{\textbf{\textit{#1}}}}
|
|
69
|
+
\newcommand{\KeywordTok}[1]{\textcolor[rgb]{0.13,0.29,0.53}{\textbf{#1}}}
|
|
70
|
+
\newcommand{\NormalTok}[1]{#1}
|
|
71
|
+
\newcommand{\OperatorTok}[1]{\textcolor[rgb]{0.81,0.36,0.00}{\textbf{#1}}}
|
|
72
|
+
\newcommand{\OtherTok}[1]{\textcolor[rgb]{0.56,0.35,0.01}{#1}}
|
|
73
|
+
\newcommand{\PreprocessorTok}[1]{\textcolor[rgb]{0.56,0.35,0.01}{\textit{#1}}}
|
|
74
|
+
\newcommand{\RegionMarkerTok}[1]{#1}
|
|
75
|
+
\newcommand{\SpecialCharTok}[1]{\textcolor[rgb]{0.81,0.36,0.00}{\textbf{#1}}}
|
|
76
|
+
\newcommand{\SpecialStringTok}[1]{\textcolor[rgb]{0.31,0.60,0.02}{#1}}
|
|
77
|
+
\newcommand{\StringTok}[1]{\textcolor[rgb]{0.31,0.60,0.02}{#1}}
|
|
78
|
+
\newcommand{\VariableTok}[1]{\textcolor[rgb]{0.00,0.00,0.00}{#1}}
|
|
79
|
+
\newcommand{\VerbatimStringTok}[1]{\textcolor[rgb]{0.31,0.60,0.02}{#1}}
|
|
80
|
+
\newcommand{\WarningTok}[1]{\textcolor[rgb]{0.56,0.35,0.01}{\textbf{\textit{#1}}}}
|
|
81
|
+
\usepackage{graphicx}
|
|
82
|
+
\makeatletter
|
|
83
|
+
\newsavebox\pandoc@box
|
|
84
|
+
\newcommand*\pandocbounded[1]{% scales image to fit in text height/width
|
|
85
|
+
\sbox\pandoc@box{#1}%
|
|
86
|
+
\Gscale@div\@tempa{\textheight}{\dimexpr\ht\pandoc@box+\dp\pandoc@box\relax}%
|
|
87
|
+
\Gscale@div\@tempb{\linewidth}{\wd\pandoc@box}%
|
|
88
|
+
\ifdim\@tempb\p@<\@tempa\p@\let\@tempa\@tempb\fi% select the smaller of both
|
|
89
|
+
\ifdim\@tempa\p@<\p@\scalebox{\@tempa}{\usebox\pandoc@box}%
|
|
90
|
+
\else\usebox{\pandoc@box}%
|
|
91
|
+
\fi%
|
|
92
|
+
}
|
|
93
|
+
% Set default figure placement to htbp
|
|
94
|
+
\def\fps@figure{htbp}
|
|
95
|
+
\makeatother
|
|
96
|
+
\setlength{\emergencystretch}{3em} % prevent overfull lines
|
|
97
|
+
\providecommand{\tightlist}{%
|
|
98
|
+
\setlength{\itemsep}{0pt}\setlength{\parskip}{0pt}}
|
|
99
|
+
% usar portugues do Brasil
|
|
100
|
+
% \usepackage[brazilian]{babel}
|
|
101
|
+
\usepackage[utf8]{inputenc}
|
|
102
|
+
|
|
103
|
+
\usepackage{geometry}
|
|
104
|
+
\geometry{a4paper, top=1in}
|
|
105
|
+
|
|
106
|
+
% needed for kableExtra
|
|
107
|
+
\usepackage{longtable}
|
|
108
|
+
\usepackage{multirow}
|
|
109
|
+
\usepackage[table]{xcolor}
|
|
110
|
+
\usepackage{wrapfig}
|
|
111
|
+
\usepackage{float}
|
|
112
|
+
\usepackage{colortbl}
|
|
113
|
+
\usepackage{pdflscape}
|
|
114
|
+
\usepackage{tabu}
|
|
115
|
+
\usepackage{threeparttable}
|
|
116
|
+
\usepackage[normalem]{ulem}
|
|
117
|
+
|
|
118
|
+
\usepackage{bbm}
|
|
119
|
+
\usepackage{booktabs}
|
|
120
|
+
\usepackage{expex}
|
|
121
|
+
|
|
122
|
+
\usepackage{graphicx}
|
|
123
|
+
|
|
124
|
+
\usepackage{fancyhdr}
|
|
125
|
+
% set the header and foot style
|
|
126
|
+
% style 'fancy' adds the section name on the header
|
|
127
|
+
% and the page number on the footer
|
|
128
|
+
\pagestyle{fancy}
|
|
129
|
+
|
|
130
|
+
% style 'fancyhf' leaves header and footer empty
|
|
131
|
+
%\fancyhf{}
|
|
132
|
+
|
|
133
|
+
% sets the left head element to \rightmark, which contains the
|
|
134
|
+
% current section (\leftmark is the current chapter)
|
|
135
|
+
%\fancyhead[L]{\rightmark} .
|
|
136
|
+
|
|
137
|
+
% sets the right head element to the page number.
|
|
138
|
+
% \fancyhead[R]{\thepage}
|
|
139
|
+
|
|
140
|
+
% lets the head rule disappear.
|
|
141
|
+
% \renewcommand{\headrulewidth}{0pt}
|
|
142
|
+
% Possible selectors for the optional argument of \fancyhead/\fancyfoot
|
|
143
|
+
% are L (left), C (center) or R (right) for the position of the element
|
|
144
|
+
% and E (even) or O (odd) to distinguish even and odd pages. If you omit
|
|
145
|
+
% E/O the element is set for all pages.
|
|
146
|
+
|
|
147
|
+
% \usepackage{lipsum}
|
|
148
|
+
|
|
149
|
+
% make available command lastpage
|
|
150
|
+
\usepackage{lastpage}
|
|
151
|
+
|
|
152
|
+
% default fontsize 11pt better to add
|
|
153
|
+
% fontsize on the yaml header
|
|
154
|
+
% \usepackage[fontsize=11pt]{scrextend}
|
|
155
|
+
|
|
156
|
+
% comandos para formatar uma tabela
|
|
157
|
+
\usepackage{array}
|
|
158
|
+
\newcolumntype{L}[1]{>{\raggedright\let\newline\\\arraybackslash\hspace{0pt}}m{#1}}
|
|
159
|
+
\newcolumntype{C}[1]{>{\centering\let\newline\\\arraybackslash\hspace{0pt}}m{#1}}
|
|
160
|
+
\newcolumntype{R}[1]{>{\raggedleft\let\newline\\\arraybackslash\hspace{0pt}}m{#1}}
|
|
161
|
+
|
|
162
|
+
% necessário if we need to import other latex documents
|
|
163
|
+
\usepackage{import}
|
|
164
|
+
|
|
165
|
+
% Command to import an R variable to latex
|
|
166
|
+
\newcommand{\RtoLatex}[2]{\newcommand{#1}{#2}}
|
|
167
|
+
|
|
168
|
+
% Soft-wrap Pandoc highlighted code/output boxes (Shaded + Highlighting).
|
|
169
|
+
% This runs after Pandoc's default \DefineVerbatimEnvironment{Highlighting}
|
|
170
|
+
% (header includes come later in the generated .tex). Prefer manual line
|
|
171
|
+
% breaks in Rmd sources; breaklines is a safety net for leftovers/output.
|
|
172
|
+
% Requires TinyTeX/TeX Live package: tlmgr install fvextra
|
|
173
|
+
\IfFileExists{fvextra.sty}{%
|
|
174
|
+
\usepackage{fvextra}%
|
|
175
|
+
\DefineVerbatimEnvironment{Highlighting}{Verbatim}{%
|
|
176
|
+
breaklines=true,
|
|
177
|
+
breakanywhere=true,
|
|
178
|
+
breakindent=1.5em,
|
|
179
|
+
fontsize=\small,
|
|
180
|
+
commandchars=\\\{\}%
|
|
181
|
+
}%
|
|
182
|
+
}{%
|
|
183
|
+
% fancyvrb is already loaded by Pandoc; shrink code so more fits per line.
|
|
184
|
+
\DefineVerbatimEnvironment{Highlighting}{Verbatim}{%
|
|
185
|
+
fontsize=\small,
|
|
186
|
+
commandchars=\\\{\}%
|
|
187
|
+
}%
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
%
|
|
191
|
+
%\newcommand{\atraso}[1]{\color{red} \textbf {Tempo desde a Assinatura do Contrato: #1 dias}}
|
|
192
|
+
\usepackage{bookmark}
|
|
193
|
+
\IfFileExists{xurl.sty}{\usepackage{xurl}}{} % add URL line breaks if available
|
|
194
|
+
\urlstyle{same}
|
|
195
|
+
\hypersetup{
|
|
196
|
+
pdftitle={Non Standard Evaluation in dplyr with Galaaz},
|
|
197
|
+
pdfauthor={Rodrigo Botafogo; Daniel Mossé - University of Pittsburgh},
|
|
198
|
+
hidelinks,
|
|
199
|
+
pdfcreator={LaTeX via pandoc}}
|
|
200
|
+
|
|
201
|
+
\title{Non Standard Evaluation in dplyr with Galaaz}
|
|
202
|
+
\author{Rodrigo Botafogo \and Daniel Mossé - University of Pittsburgh}
|
|
203
|
+
\date{10/05/2019 (narrative updated for Galaaz 2.0, 2026)}
|
|
204
|
+
|
|
205
|
+
\begin{document}
|
|
206
|
+
\maketitle
|
|
207
|
+
|
|
208
|
+
{
|
|
209
|
+
\setcounter{tocdepth}{2}
|
|
210
|
+
\tableofcontents
|
|
211
|
+
}
|
|
212
|
+
\section{Introduction}\label{introduction}
|
|
213
|
+
|
|
214
|
+
According to Steven Sagaert's answer on Quora about ``Is programming
|
|
215
|
+
language R overrated?'':
|
|
216
|
+
|
|
217
|
+
\begin{quote}
|
|
218
|
+
R is a sophisticated language with an unusual (i.e.~non-mainstream) set
|
|
219
|
+
of features. It`s an impure functional programming language with
|
|
220
|
+
sophisticated metaprogramming and 3 different OO systems.
|
|
221
|
+
\end{quote}
|
|
222
|
+
|
|
223
|
+
\begin{quote}
|
|
224
|
+
Just like common lisp you can completely customise how things work via
|
|
225
|
+
metaprogramming. The biggest example is the tidyverse: by creating it's
|
|
226
|
+
own evaluation system (tidyeval) was able to create a custom syntax for
|
|
227
|
+
dplyr.
|
|
228
|
+
\end{quote}
|
|
229
|
+
|
|
230
|
+
\begin{quote}
|
|
231
|
+
Mastering R (the language) and its ecosystem is not a matter of weeks or
|
|
232
|
+
months but takes years. The rabbit hole goes pretty deep\ldots{}
|
|
233
|
+
\end{quote}
|
|
234
|
+
|
|
235
|
+
Although a highly configurable language can give programmers a great
|
|
236
|
+
deal of power, it can also take years to master---as noted above.
|
|
237
|
+
Programming with \emph{dplyr}, for instance, means learning evaluation
|
|
238
|
+
rules that are not always approachable for \textbf{statisticians and
|
|
239
|
+
analysts who are not full-time software engineers}. That is not a
|
|
240
|
+
criticism: R was \textbf{built} for \textbf{statisticians} who need
|
|
241
|
+
trustworthy results on a deadline, not necessarily for building large
|
|
242
|
+
applications.
|
|
243
|
+
|
|
244
|
+
\textbf{Unfortunately}, when such a user moves on to more
|
|
245
|
+
\textbf{sophisticated} programming patterns, the learning curve can
|
|
246
|
+
become a real hurdle.
|
|
247
|
+
|
|
248
|
+
In this post we will see how to program with \emph{dplyr} in Galaaz and
|
|
249
|
+
how Ruby can simplify the learning curve of mastering \emph{dplyr}
|
|
250
|
+
coding.
|
|
251
|
+
|
|
252
|
+
\section{But first, what is Galaaz??}\label{but-first-what-is-galaaz}
|
|
253
|
+
|
|
254
|
+
Galaaz is a system for tightly coupling Ruby and R. Ruby is a powerful
|
|
255
|
+
language, with a large community, a very large set of libraries and
|
|
256
|
+
great for web development. It is also easy to learn. However, it lacks
|
|
257
|
+
libraries for data science, statistics, scientific plotting and machine
|
|
258
|
+
learning. On the other hand, R is considered one of the most powerful
|
|
259
|
+
languages for solving all of the above problems. \textbf{Python} is a
|
|
260
|
+
strong competitor, with NumPy, pandas, SciPy, scikit-learn, and
|
|
261
|
+
\textbf{many thousands} of other packages on PyPI. We will not dwell on
|
|
262
|
+
R \textbf{versus} Python here: both are excellent languages with
|
|
263
|
+
different strengths. Our interest is to bring to yet another excellent
|
|
264
|
+
language, Ruby, the data science libraries that it lacks.
|
|
265
|
+
|
|
266
|
+
With Galaaz we do not intend to re-implement any of the scientific
|
|
267
|
+
libraries in R. However, we allow for very tight coupling between the
|
|
268
|
+
two languages to the point that the Ruby developer does not need to know
|
|
269
|
+
that there is an R engine running. Also, from the point of view of the R
|
|
270
|
+
user/developer, Galaaz looks a lot like R, with just minor syntactic
|
|
271
|
+
difference, so there is almost no learning curve for the R developer.
|
|
272
|
+
And as we will see in this post that programming with \emph{dplyr} is
|
|
273
|
+
easier in Galaaz than in R.
|
|
274
|
+
|
|
275
|
+
R users are probably quite knowledgeable about \emph{dplyr}. For the
|
|
276
|
+
Ruby developer, \emph{dplyr} and the \emph{tidyverse} libraries are a
|
|
277
|
+
set of libraries for data manipulation in R, developed by Hadley
|
|
278
|
+
Wickham, Chief Scientist at Posit (formerly RStudio) and a prolific R
|
|
279
|
+
coder and writer.
|
|
280
|
+
|
|
281
|
+
For the coupling of Ruby and R, \textbf{Galaaz 2.0} uses
|
|
282
|
+
\textbf{\href{https://www.jruby.org/}{JRuby}} (Ruby on the JVM) together
|
|
283
|
+
with \textbf{GNU R}. A \textbf{bridge} sends expressions and data
|
|
284
|
+
between Ruby and an R process so that Ruby can call \textbf{dplyr} and
|
|
285
|
+
the rest of the tidyverse as if they were part of the same workflow. An
|
|
286
|
+
\textbf{earlier} Galaaz line of work used Oracle's \textbf{GraalVM} with
|
|
287
|
+
\textbf{TruffleRuby} and \textbf{FastR} in a single runtime; that
|
|
288
|
+
approach is \textbf{no longer} the supported stack---see the project
|
|
289
|
+
\textbf{manual} for setup, \textbf{\texttt{bin/galaaz-jruby}}, and
|
|
290
|
+
\textbf{gKnit}.
|
|
291
|
+
|
|
292
|
+
\section{Tidyverse and dplyr}\label{tidyverse-and-dplyr}
|
|
293
|
+
|
|
294
|
+
In
|
|
295
|
+
\href{https://rviews.rstudio.com/2017/06/08/what-is-the-tidyverse/}{What
|
|
296
|
+
is the tidyverse?} the tidyverse is explained as follows:
|
|
297
|
+
|
|
298
|
+
\begin{quote}
|
|
299
|
+
The tidyverse is a coherent system of packages for data manipulation,
|
|
300
|
+
exploration and visualization that share a common design philosophy.
|
|
301
|
+
These were mostly developed by Hadley Wickham himself, but they are now
|
|
302
|
+
being expanded by several contributors. Tidyverse packages are intended
|
|
303
|
+
to make statisticians and data scientists more productive by guiding
|
|
304
|
+
them through workflows that facilitate communication, and result in
|
|
305
|
+
reproducible work products. Fundamentally, the tidyverse is about the
|
|
306
|
+
connections between the tools that make the workflow possible.
|
|
307
|
+
\end{quote}
|
|
308
|
+
|
|
309
|
+
\emph{dplyr} is one of the many packages that are part of the tidyverse.
|
|
310
|
+
It is:
|
|
311
|
+
|
|
312
|
+
\begin{quote}
|
|
313
|
+
a grammar of data manipulation, providing a consistent set of verbs that
|
|
314
|
+
help you solve the most common data manipulation challenges:
|
|
315
|
+
\end{quote}
|
|
316
|
+
|
|
317
|
+
\begin{quote}
|
|
318
|
+
\begin{enumerate}
|
|
319
|
+
\def\labelenumi{\arabic{enumi}.}
|
|
320
|
+
\tightlist
|
|
321
|
+
\item
|
|
322
|
+
mutate() adds new variables that are functions of existing variables
|
|
323
|
+
\item
|
|
324
|
+
select() picks variables based on their names.
|
|
325
|
+
\item
|
|
326
|
+
filter() picks cases based on their values.
|
|
327
|
+
\item
|
|
328
|
+
summarise() reduces multiple values down to a single summary.
|
|
329
|
+
\item
|
|
330
|
+
arrange() changes the ordering of the rows.
|
|
331
|
+
\end{enumerate}
|
|
332
|
+
\end{quote}
|
|
333
|
+
|
|
334
|
+
Very often R is used interactively and users use \emph{dplyr} to
|
|
335
|
+
manipulate a single dataset without programming. When users want to
|
|
336
|
+
replicate their work for multiple datasets, programming becomes
|
|
337
|
+
necessary.
|
|
338
|
+
|
|
339
|
+
\section{Programming with dplyr}\label{programming-with-dplyr}
|
|
340
|
+
|
|
341
|
+
In the vignette
|
|
342
|
+
\href{https://dplyr.tidyverse.org/articles/programming.html}{``Programming
|
|
343
|
+
with dplyr''}, Hadley Wickham states:
|
|
344
|
+
|
|
345
|
+
\begin{quote}
|
|
346
|
+
Most dplyr functions use non-standard evaluation (NSE). This is a
|
|
347
|
+
catch-all term that means they don't follow the usual R rules of
|
|
348
|
+
evaluation. Instead, they capture the expression that you typed and
|
|
349
|
+
evaluate it in a custom way. This has two main benefits for dplyr code:
|
|
350
|
+
\end{quote}
|
|
351
|
+
|
|
352
|
+
\begin{quote}
|
|
353
|
+
Operations on data frames can be expressed succinctly because you don't
|
|
354
|
+
need to repeat the name of the data frame. For example, you can write
|
|
355
|
+
filter(df, x == 1, y == 2, z == 3) instead of df{[}df\$x == 1 \& df\$y
|
|
356
|
+
==2 \& df\$z == 3, {]}.
|
|
357
|
+
\end{quote}
|
|
358
|
+
|
|
359
|
+
\begin{quote}
|
|
360
|
+
dplyr can choose to compute results in a different way to base R. This
|
|
361
|
+
is important for database backends because dplyr itself doesn't do any
|
|
362
|
+
work, but instead generates the SQL that tells the database what to do.
|
|
363
|
+
\end{quote}
|
|
364
|
+
|
|
365
|
+
But then he goes on:
|
|
366
|
+
|
|
367
|
+
\begin{quote}
|
|
368
|
+
Unfortunately these benefits do not come for free. There are two main
|
|
369
|
+
drawbacks:
|
|
370
|
+
\end{quote}
|
|
371
|
+
|
|
372
|
+
\begin{quote}
|
|
373
|
+
Most dplyr arguments are not referentially transparent. That means you
|
|
374
|
+
can't replace a value with a seemingly equivalent object that you've
|
|
375
|
+
defined elsewhere. In other words, this code:
|
|
376
|
+
\end{quote}
|
|
377
|
+
|
|
378
|
+
\begin{Shaded}
|
|
379
|
+
\begin{Highlighting}[]
|
|
380
|
+
\NormalTok{df }\OtherTok{\textless{}{-}} \FunctionTok{data.frame}\NormalTok{(}\AttributeTok{x =} \DecValTok{1}\SpecialCharTok{:}\DecValTok{3}\NormalTok{, }\AttributeTok{y =} \DecValTok{3}\SpecialCharTok{:}\DecValTok{1}\NormalTok{)}
|
|
381
|
+
\FunctionTok{print}\NormalTok{(}\FunctionTok{filter}\NormalTok{(df, x }\SpecialCharTok{==} \DecValTok{1}\NormalTok{))}
|
|
382
|
+
\CommentTok{\#\textgreater{} \# A tibble: 1 x 2}
|
|
383
|
+
\CommentTok{\#\textgreater{} x y}
|
|
384
|
+
\CommentTok{\#\textgreater{} \textless{}int\textgreater{} \textless{}int\textgreater{}}
|
|
385
|
+
\CommentTok{\#\textgreater{} 1 1 3}
|
|
386
|
+
\end{Highlighting}
|
|
387
|
+
\end{Shaded}
|
|
388
|
+
|
|
389
|
+
\begin{quote}
|
|
390
|
+
Is not equivalent to this code:
|
|
391
|
+
\end{quote}
|
|
392
|
+
|
|
393
|
+
\begin{Shaded}
|
|
394
|
+
\begin{Highlighting}[]
|
|
395
|
+
\NormalTok{my\_var }\OtherTok{\textless{}{-}}\NormalTok{ x}
|
|
396
|
+
\CommentTok{\#\textgreater{} Error in eval(expr, envir, enclos): object \textquotesingle{}x\textquotesingle{} not found}
|
|
397
|
+
\FunctionTok{filter}\NormalTok{(df, my\_var }\SpecialCharTok{==} \DecValTok{1}\NormalTok{)}
|
|
398
|
+
\CommentTok{\#\textgreater{} Error: object \textquotesingle{}my\_var\textquotesingle{} not found}
|
|
399
|
+
\end{Highlighting}
|
|
400
|
+
\end{Shaded}
|
|
401
|
+
|
|
402
|
+
\begin{quote}
|
|
403
|
+
This makes it hard to create functions with arguments that change how
|
|
404
|
+
dplyr verbs are computed.
|
|
405
|
+
\end{quote}
|
|
406
|
+
|
|
407
|
+
As a result of this, programming with \emph{dplyr} requires learning a
|
|
408
|
+
set of new ideas and concepts. In this vignette Hadley goes on showing
|
|
409
|
+
how to program ever more difficult problems with \emph{dplyr}, showing
|
|
410
|
+
the problems it faces and the new concepts needed to solve them.
|
|
411
|
+
|
|
412
|
+
In this blog, we will look at all the problems presented by Harley on
|
|
413
|
+
the vignette and show how those same problems can be solved using Galaaz
|
|
414
|
+
and the Ruby language.
|
|
415
|
+
|
|
416
|
+
This blog is organized as follows: first we show how to write
|
|
417
|
+
expressions using Galaaz.\\
|
|
418
|
+
Expressions are a fundamental concept in \emph{dplyr} and are not part
|
|
419
|
+
of basic Ruby. We extend the Ruby language create a manipulate
|
|
420
|
+
expressions that will be used by \emph{dplyr} functions.
|
|
421
|
+
|
|
422
|
+
Then we show very succintly how Ruby and R can be integrated and how R
|
|
423
|
+
functions are transparently called from Ruby. Galaaz
|
|
424
|
+
\href{https://github.com/rbotafogo/galaaz/wiki}{user manual} (still in
|
|
425
|
+
development) goes in much deeper detail about this integration.
|
|
426
|
+
|
|
427
|
+
Next in section ``Data manipulation wiht \emph{dplyr}'' we go through
|
|
428
|
+
all the problems on the \emph{dplyr} vignette and look at how they are
|
|
429
|
+
solved in Galaaz. We then discuss why programming with Galaaz and
|
|
430
|
+
\emph{dplyr} is easier than programming with \emph{dplyr} in plain R.
|
|
431
|
+
|
|
432
|
+
The following section looks at another more advanced problem and shows
|
|
433
|
+
that Galaaz can still handle it without any difficulty. We then provide
|
|
434
|
+
further reading and concluding remarks.
|
|
435
|
+
|
|
436
|
+
\section{Writing Expressions in
|
|
437
|
+
Galaaz}\label{writing-expressions-in-galaaz}
|
|
438
|
+
|
|
439
|
+
Galaaz extends Ruby to work with expressions, similar to R's expressions
|
|
440
|
+
build with `quote' (base R) or `quo' (tidyverse). Expressions in this
|
|
441
|
+
context are like mathematical expressions or formulae. For instance, in
|
|
442
|
+
mathematics, the expression \(y = sin(x)\) describes a function but
|
|
443
|
+
cannot be computed unless the value of \(x\) is bound to some value.
|
|
444
|
+
|
|
445
|
+
Expressions are fundamental in \emph{dplyr} programming as they are the
|
|
446
|
+
input to \emph{dplyr} functions, for instance, as we will see shortly,
|
|
447
|
+
if a data frame has a column named `x' and we want to add another
|
|
448
|
+
column, y, to this dataframe that has the values of `x' times 2, then we
|
|
449
|
+
would call a \emph{dplyr} function with the expression `y = x * 2'.
|
|
450
|
+
|
|
451
|
+
\subsection{A note on notation}\label{a-note-on-notation}
|
|
452
|
+
|
|
453
|
+
This blog was written in Rmarkdown and automatically converted to HTML
|
|
454
|
+
or PDF (depending on where you are reading this blog) with gKnit (a tool
|
|
455
|
+
provided by Galaaz). In Rmarkdown, it is possible to write text and code
|
|
456
|
+
blocks that are executed to generate the final report. Code blocks
|
|
457
|
+
appear inside a `box' and the result of their execution appear either in
|
|
458
|
+
another type of `box' with a different background (HTML) or as normal
|
|
459
|
+
text (PDF). Every output line from the code execution is preceded by
|
|
460
|
+
`\#\#'.
|
|
461
|
+
|
|
462
|
+
\subsection{Expressions from
|
|
463
|
+
operators}\label{expressions-from-operators}
|
|
464
|
+
|
|
465
|
+
The code below creates an expression summing two symbols. Note that :a
|
|
466
|
+
and :b are Ruby symbols and are not bound to any values at the time of
|
|
467
|
+
expression definition:
|
|
468
|
+
|
|
469
|
+
\begin{Shaded}
|
|
470
|
+
\begin{Highlighting}[]
|
|
471
|
+
\ControlFlowTok{begin}
|
|
472
|
+
\NormalTok{ exp1 }\OperatorTok{=} \WarningTok{:a} \OperatorTok{+} \WarningTok{:b}
|
|
473
|
+
\FunctionTok{puts}\NormalTok{ exp1}
|
|
474
|
+
\ControlFlowTok{rescue} \OperatorTok{=\textgreater{}}\NormalTok{ e}
|
|
475
|
+
\CommentTok{\# Bare Symbol\#+ is not expression sugar in Galaaz 2.0; short error for PDF.}
|
|
476
|
+
\FunctionTok{puts} \StringTok{"}\SpecialCharTok{\#\{}\NormalTok{e}\AttributeTok{.class}\SpecialCharTok{\}}\StringTok{: }\SpecialCharTok{\#\{}\NormalTok{e}\AttributeTok{.message}\SpecialCharTok{\}}\StringTok{"}
|
|
477
|
+
\ControlFlowTok{end}
|
|
478
|
+
\end{Highlighting}
|
|
479
|
+
\end{Shaded}
|
|
480
|
+
|
|
481
|
+
\begin{verbatim}
|
|
482
|
+
## NoMethodError: undefined method '+' for an instance of Symbol
|
|
483
|
+
\end{verbatim}
|
|
484
|
+
|
|
485
|
+
In Galaaz, we can build any complex mathematical expression such as:
|
|
486
|
+
|
|
487
|
+
\begin{Shaded}
|
|
488
|
+
\begin{Highlighting}[]
|
|
489
|
+
\NormalTok{exp2 }\OperatorTok{=}\NormalTok{ (}\ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:b}\KeywordTok{]}\NormalTok{) }\OperatorTok{*} \FloatTok{2.0} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:c}\KeywordTok{]} \OperatorTok{**} \DecValTok{2} \OperatorTok{/} \ConstantTok{R}\KeywordTok{[}\WarningTok{:z}\KeywordTok{]}
|
|
490
|
+
\FunctionTok{puts}\NormalTok{ exp2}
|
|
491
|
+
\end{Highlighting}
|
|
492
|
+
\end{Shaded}
|
|
493
|
+
|
|
494
|
+
\begin{verbatim}
|
|
495
|
+
## a + b * 2.0 + c ^ 2L / z
|
|
496
|
+
\end{verbatim}
|
|
497
|
+
|
|
498
|
+
Expressions are printed with the same format as the equivalent R
|
|
499
|
+
expressions. The `L' after 2 indicates that 2 is an integer.
|
|
500
|
+
|
|
501
|
+
The R developer should note that in R, if she writes the number `2', the
|
|
502
|
+
R interpreter will convert it to float. In order to get an interger she
|
|
503
|
+
should write `2L'. Galaaz follows Ruby notation and `2' is an integer,
|
|
504
|
+
while `2.0' is a float.
|
|
505
|
+
|
|
506
|
+
It is also possible to use inequality operators in building expressions:
|
|
507
|
+
|
|
508
|
+
\begin{Shaded}
|
|
509
|
+
\begin{Highlighting}[]
|
|
510
|
+
\NormalTok{exp3 }\OperatorTok{=}\NormalTok{ (}\ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:b}\KeywordTok{]}\NormalTok{) }\OperatorTok{\textgreater{}=} \ConstantTok{R}\KeywordTok{[}\WarningTok{:z}\KeywordTok{]}
|
|
511
|
+
\FunctionTok{puts}\NormalTok{ exp3}
|
|
512
|
+
\end{Highlighting}
|
|
513
|
+
\end{Shaded}
|
|
514
|
+
|
|
515
|
+
\begin{verbatim}
|
|
516
|
+
## a + b >= z
|
|
517
|
+
\end{verbatim}
|
|
518
|
+
|
|
519
|
+
Expressions' definition can also make use of normal Ruby variables
|
|
520
|
+
without any problem:
|
|
521
|
+
|
|
522
|
+
\begin{Shaded}
|
|
523
|
+
\begin{Highlighting}[]
|
|
524
|
+
\NormalTok{x }\OperatorTok{=} \DecValTok{20}
|
|
525
|
+
\NormalTok{y }\OperatorTok{=} \FloatTok{30.0}
|
|
526
|
+
\NormalTok{exp\_var }\OperatorTok{=}\NormalTok{ (}\ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:b}\KeywordTok{]}\NormalTok{) }\OperatorTok{*}\NormalTok{ x }\OperatorTok{\textless{}=} \ConstantTok{R}\KeywordTok{[}\WarningTok{:z}\KeywordTok{]} \OperatorTok{{-}}\NormalTok{ y}
|
|
527
|
+
\FunctionTok{puts}\NormalTok{ exp\_var}
|
|
528
|
+
\end{Highlighting}
|
|
529
|
+
\end{Shaded}
|
|
530
|
+
|
|
531
|
+
\begin{verbatim}
|
|
532
|
+
## a + b * 20L <= z - 30.0
|
|
533
|
+
\end{verbatim}
|
|
534
|
+
|
|
535
|
+
Galaaz provides both symbolic representations for operators, such as
|
|
536
|
+
(\textgreater, \textless, !=) as functional notation for those operators
|
|
537
|
+
such as (.gt, .ge, etc.). So the same expression written above can also
|
|
538
|
+
be written as
|
|
539
|
+
|
|
540
|
+
\begin{Shaded}
|
|
541
|
+
\begin{Highlighting}[]
|
|
542
|
+
\NormalTok{exp4 }\OperatorTok{=}\NormalTok{ (}\ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:b}\KeywordTok{]}\NormalTok{)}\AttributeTok{.ge} \ConstantTok{R}\KeywordTok{[}\WarningTok{:z}\KeywordTok{]}
|
|
543
|
+
\FunctionTok{puts}\NormalTok{ exp4}
|
|
544
|
+
\end{Highlighting}
|
|
545
|
+
\end{Shaded}
|
|
546
|
+
|
|
547
|
+
\begin{verbatim}
|
|
548
|
+
## a + b >= z
|
|
549
|
+
\end{verbatim}
|
|
550
|
+
|
|
551
|
+
Two types of expressions, however, can only be created with the
|
|
552
|
+
functional representation of the operators. Those are expressions
|
|
553
|
+
involving `==', and `='. This is the case since those symbols have
|
|
554
|
+
special meaning in Ruby and should not be redefined.
|
|
555
|
+
|
|
556
|
+
In order to write an expression involving `==' we need to use the method
|
|
557
|
+
`.eq' and for `=' we need the function `.assign':
|
|
558
|
+
|
|
559
|
+
\begin{Shaded}
|
|
560
|
+
\begin{Highlighting}[]
|
|
561
|
+
\NormalTok{exp5 }\OperatorTok{=}\NormalTok{ (}\ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:b}\KeywordTok{]}\NormalTok{)}\AttributeTok{.eq} \ConstantTok{R}\KeywordTok{[}\WarningTok{:z}\KeywordTok{]}
|
|
562
|
+
\FunctionTok{puts}\NormalTok{ exp5}
|
|
563
|
+
\end{Highlighting}
|
|
564
|
+
\end{Shaded}
|
|
565
|
+
|
|
566
|
+
\begin{verbatim}
|
|
567
|
+
## a + b == z
|
|
568
|
+
\end{verbatim}
|
|
569
|
+
|
|
570
|
+
\begin{Shaded}
|
|
571
|
+
\begin{Highlighting}[]
|
|
572
|
+
\NormalTok{exp6 }\OperatorTok{=} \ConstantTok{R}\KeywordTok{[}\WarningTok{:y}\KeywordTok{]}\AttributeTok{.assign} \ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:b}\KeywordTok{]}
|
|
573
|
+
\FunctionTok{puts}\NormalTok{ exp6}
|
|
574
|
+
\end{Highlighting}
|
|
575
|
+
\end{Shaded}
|
|
576
|
+
|
|
577
|
+
\begin{verbatim}
|
|
578
|
+
## y <- a + b
|
|
579
|
+
\end{verbatim}
|
|
580
|
+
|
|
581
|
+
Users should be careful when writing expressions not to inadvertently
|
|
582
|
+
use `==' or `=' as this will generate an error, that might be a bit
|
|
583
|
+
cryptic (in future releases of Galaza, we plan to improve the error
|
|
584
|
+
message).
|
|
585
|
+
|
|
586
|
+
\begin{Shaded}
|
|
587
|
+
\begin{Highlighting}[]
|
|
588
|
+
\NormalTok{exp\_wrong }\OperatorTok{=}\NormalTok{ (}\ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:b}\KeywordTok{]}\NormalTok{) }\OperatorTok{==} \ConstantTok{R}\KeywordTok{[}\WarningTok{:z}\KeywordTok{]}
|
|
589
|
+
\FunctionTok{puts}\NormalTok{ exp\_wrong}
|
|
590
|
+
\end{Highlighting}
|
|
591
|
+
\end{Shaded}
|
|
592
|
+
|
|
593
|
+
\begin{verbatim}
|
|
594
|
+
## false
|
|
595
|
+
\end{verbatim}
|
|
596
|
+
|
|
597
|
+
The problem lies with the fact that when using `==' we are comparing
|
|
598
|
+
expression (R{[}:a{]} + R{[}:b{]}) to expression R{[}:z{]} with `=='.
|
|
599
|
+
When this comparison is executed, the system tries to evaluate :a, :b
|
|
600
|
+
and :z, and those symbols, at this time, are not bound to anything
|
|
601
|
+
giving the ``object `a' not found'' message.
|
|
602
|
+
|
|
603
|
+
\subsection{Expressions with R
|
|
604
|
+
methods}\label{expressions-with-r-methods}
|
|
605
|
+
|
|
606
|
+
It is often necessary to create an expression that uses a method or
|
|
607
|
+
function. For instance, in mathematics, it's quite natural to write an
|
|
608
|
+
expressin such as \(y = sin(x)\). In this case, the `sin' function is
|
|
609
|
+
part of the expression and should not be immediately executed. When we
|
|
610
|
+
want the function to be part of the expression, we call the function
|
|
611
|
+
preceeding it by the letter E, such as `E.sin(x)'
|
|
612
|
+
|
|
613
|
+
\begin{Shaded}
|
|
614
|
+
\begin{Highlighting}[]
|
|
615
|
+
\NormalTok{exp7 }\OperatorTok{=} \ConstantTok{R}\KeywordTok{[}\WarningTok{:y}\KeywordTok{]}\AttributeTok{.assign} \ConstantTok{E}\AttributeTok{.sin}\NormalTok{(}\ConstantTok{R}\KeywordTok{[}\WarningTok{:x}\KeywordTok{]}\NormalTok{)}
|
|
616
|
+
\FunctionTok{puts}\NormalTok{ exp7}
|
|
617
|
+
\end{Highlighting}
|
|
618
|
+
\end{Shaded}
|
|
619
|
+
|
|
620
|
+
\begin{verbatim}
|
|
621
|
+
## y <- sin(x)
|
|
622
|
+
\end{verbatim}
|
|
623
|
+
|
|
624
|
+
Function expressions can also be written using `.' notation:
|
|
625
|
+
|
|
626
|
+
\begin{Shaded}
|
|
627
|
+
\begin{Highlighting}[]
|
|
628
|
+
\NormalTok{exp8 }\OperatorTok{=} \ConstantTok{R}\KeywordTok{[}\WarningTok{:y}\KeywordTok{]}\AttributeTok{.assign} \ConstantTok{R}\KeywordTok{[}\WarningTok{:x}\KeywordTok{]}\AttributeTok{.sin}
|
|
629
|
+
\FunctionTok{puts}\NormalTok{ exp8}
|
|
630
|
+
\end{Highlighting}
|
|
631
|
+
\end{Shaded}
|
|
632
|
+
|
|
633
|
+
\begin{verbatim}
|
|
634
|
+
## y <- sin(x)
|
|
635
|
+
\end{verbatim}
|
|
636
|
+
|
|
637
|
+
When a function has multiple arguments, the first one can be used before
|
|
638
|
+
the `.'. For instance, the R concatenate function `c', that concatenates
|
|
639
|
+
two or more arguments can be part of an expression as:
|
|
640
|
+
|
|
641
|
+
\begin{Shaded}
|
|
642
|
+
\begin{Highlighting}[]
|
|
643
|
+
\NormalTok{exp9 }\OperatorTok{=} \ConstantTok{R}\KeywordTok{[}\WarningTok{:x}\KeywordTok{]}\AttributeTok{.c}\NormalTok{(}\ConstantTok{R}\KeywordTok{[}\WarningTok{:y}\KeywordTok{]}\NormalTok{)}
|
|
644
|
+
\FunctionTok{puts}\NormalTok{ exp9}
|
|
645
|
+
\end{Highlighting}
|
|
646
|
+
\end{Shaded}
|
|
647
|
+
|
|
648
|
+
\begin{verbatim}
|
|
649
|
+
## c(x, y)
|
|
650
|
+
\end{verbatim}
|
|
651
|
+
|
|
652
|
+
Note that this gives an OO feeling to the code, as if we were saying `x'
|
|
653
|
+
concatenates `y'. As a side note, `.' notation can be used as the R pipe
|
|
654
|
+
operator `\%\textgreater\%', but is more general than the pipe.
|
|
655
|
+
|
|
656
|
+
\subsection{Evaluating an Expression}\label{evaluating-an-expression}
|
|
657
|
+
|
|
658
|
+
Although we are mainly focusing on expressions to pass them to
|
|
659
|
+
\emph{dplyr} functions, expressions can be evaluated by calling function
|
|
660
|
+
`eval' with a binding.
|
|
661
|
+
|
|
662
|
+
A binding can be provided with a list or a data frame as shown below:
|
|
663
|
+
|
|
664
|
+
\begin{Shaded}
|
|
665
|
+
\begin{Highlighting}[]
|
|
666
|
+
\NormalTok{exp }\OperatorTok{=}\NormalTok{ (}\ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:b}\KeywordTok{]}\NormalTok{) }\OperatorTok{*} \FloatTok{2.0} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:c}\KeywordTok{]} \OperatorTok{**} \DecValTok{2} \OperatorTok{/} \ConstantTok{R}\KeywordTok{[}\WarningTok{:z}\KeywordTok{]}
|
|
667
|
+
\FunctionTok{puts}\NormalTok{ exp}\AttributeTok{.eval}\NormalTok{(}\ConstantTok{R}\AttributeTok{.list}\NormalTok{(}\WarningTok{a:} \DecValTok{10}\NormalTok{, }\WarningTok{b:} \DecValTok{20}\NormalTok{, }\WarningTok{c:} \DecValTok{30}\NormalTok{, }\WarningTok{z:} \DecValTok{40}\NormalTok{))}
|
|
668
|
+
\end{Highlighting}
|
|
669
|
+
\end{Shaded}
|
|
670
|
+
|
|
671
|
+
\begin{verbatim}
|
|
672
|
+
## [1] 72.5
|
|
673
|
+
\end{verbatim}
|
|
674
|
+
|
|
675
|
+
with a data frame:
|
|
676
|
+
|
|
677
|
+
\begin{Shaded}
|
|
678
|
+
\begin{Highlighting}[]
|
|
679
|
+
\NormalTok{df }\OperatorTok{=} \ConstantTok{R}\AttributeTok{.data\_\_frame}\NormalTok{(}
|
|
680
|
+
\WarningTok{a:} \ConstantTok{R}\AttributeTok{.c}\NormalTok{(}\DecValTok{1}\NormalTok{, }\DecValTok{2}\NormalTok{, }\DecValTok{3}\NormalTok{),}
|
|
681
|
+
\WarningTok{b:} \ConstantTok{R}\AttributeTok{.c}\NormalTok{(}\DecValTok{10}\NormalTok{, }\DecValTok{20}\NormalTok{, }\DecValTok{30}\NormalTok{),}
|
|
682
|
+
\WarningTok{c:} \ConstantTok{R}\AttributeTok{.c}\NormalTok{(}\DecValTok{100}\NormalTok{, }\DecValTok{200}\NormalTok{, }\DecValTok{300}\NormalTok{),}
|
|
683
|
+
\WarningTok{z:} \ConstantTok{R}\AttributeTok{.c}\NormalTok{(}\DecValTok{1000}\NormalTok{, }\DecValTok{2000}\NormalTok{, }\DecValTok{3000}\NormalTok{))}
|
|
684
|
+
|
|
685
|
+
\FunctionTok{puts}\NormalTok{ exp}\AttributeTok{.eval}\NormalTok{(df)}
|
|
686
|
+
\end{Highlighting}
|
|
687
|
+
\end{Shaded}
|
|
688
|
+
|
|
689
|
+
\begin{verbatim}
|
|
690
|
+
## [1] 31 62 93
|
|
691
|
+
\end{verbatim}
|
|
692
|
+
|
|
693
|
+
\section{Using Galaaz to call R
|
|
694
|
+
functions}\label{using-galaaz-to-call-r-functions}
|
|
695
|
+
|
|
696
|
+
Galaaz tries to emulate as closely as possible the way R functions are
|
|
697
|
+
called and migrating from R to Galaaz should be quite easy requiring
|
|
698
|
+
only minor syntactic changes to an R script. In this post, we do not
|
|
699
|
+
have enough space to write a complete manual on Galaaz (a short manual
|
|
700
|
+
can be found at: \url{https://www.rubydoc.info/gems/galaaz/0.4.9}), so
|
|
701
|
+
we will present only a few examples scripts using Galaaz.
|
|
702
|
+
|
|
703
|
+
Basically, to call an R function from Ruby with Galaaz, one only needs
|
|
704
|
+
to preced the function with `R.'. For instance, to create a vector in R,
|
|
705
|
+
the `c' function is used. In Galaaz, a vector can be created by using
|
|
706
|
+
`R.c':
|
|
707
|
+
|
|
708
|
+
\begin{Shaded}
|
|
709
|
+
\begin{Highlighting}[]
|
|
710
|
+
\NormalTok{vec }\OperatorTok{=} \ConstantTok{R}\AttributeTok{.c}\NormalTok{(}\FloatTok{1.0}\NormalTok{, }\DecValTok{2}\NormalTok{, }\DecValTok{3}\NormalTok{)}
|
|
711
|
+
\FunctionTok{puts}\NormalTok{ vec}
|
|
712
|
+
\end{Highlighting}
|
|
713
|
+
\end{Shaded}
|
|
714
|
+
|
|
715
|
+
\begin{verbatim}
|
|
716
|
+
## [1] 1 2 3
|
|
717
|
+
\end{verbatim}
|
|
718
|
+
|
|
719
|
+
A list is created in R with the `list' function, so in Galaaz we do:
|
|
720
|
+
|
|
721
|
+
\begin{Shaded}
|
|
722
|
+
\begin{Highlighting}[]
|
|
723
|
+
\NormalTok{list }\OperatorTok{=} \ConstantTok{R}\AttributeTok{.list}\NormalTok{(}\WarningTok{a:} \FloatTok{1.0}\NormalTok{, }\WarningTok{b:} \DecValTok{2}\NormalTok{, }\WarningTok{c:} \DecValTok{3}\NormalTok{)}
|
|
724
|
+
\FunctionTok{puts}\NormalTok{ list}
|
|
725
|
+
\end{Highlighting}
|
|
726
|
+
\end{Shaded}
|
|
727
|
+
|
|
728
|
+
\begin{verbatim}
|
|
729
|
+
## $a
|
|
730
|
+
## [1] 1
|
|
731
|
+
##
|
|
732
|
+
## $b
|
|
733
|
+
## [1] 2
|
|
734
|
+
##
|
|
735
|
+
## $c
|
|
736
|
+
## [1] 3
|
|
737
|
+
\end{verbatim}
|
|
738
|
+
|
|
739
|
+
Note that we can use named arguments in our list. The same code in R
|
|
740
|
+
would be:
|
|
741
|
+
|
|
742
|
+
\begin{Shaded}
|
|
743
|
+
\begin{Highlighting}[]
|
|
744
|
+
\NormalTok{lst }\OtherTok{=} \FunctionTok{list}\NormalTok{(}\AttributeTok{a =} \DecValTok{1}\NormalTok{, }\AttributeTok{b =} \DecValTok{2}\DataTypeTok{L}\NormalTok{, }\AttributeTok{c =} \DecValTok{3}\DataTypeTok{L}\NormalTok{)}
|
|
745
|
+
\FunctionTok{print}\NormalTok{(lst)}
|
|
746
|
+
\end{Highlighting}
|
|
747
|
+
\end{Shaded}
|
|
748
|
+
|
|
749
|
+
\begin{verbatim}
|
|
750
|
+
## $a
|
|
751
|
+
## [1] 1
|
|
752
|
+
##
|
|
753
|
+
## $b
|
|
754
|
+
## [1] 2
|
|
755
|
+
##
|
|
756
|
+
## $c
|
|
757
|
+
## [1] 3
|
|
758
|
+
\end{verbatim}
|
|
759
|
+
|
|
760
|
+
Now, let's say that `x' is an angle of 45\(^\circ\) and we acttually
|
|
761
|
+
want to create the expression \(y = sin(45^\circ)\), which is
|
|
762
|
+
\(y = 0.850...\). In this case, we will use `R.sin':
|
|
763
|
+
|
|
764
|
+
\begin{Shaded}
|
|
765
|
+
\begin{Highlighting}[]
|
|
766
|
+
\NormalTok{exp10 }\OperatorTok{=} \ConstantTok{R}\KeywordTok{[}\WarningTok{:y}\KeywordTok{]}\AttributeTok{.assign} \ConstantTok{R}\AttributeTok{.sin}\NormalTok{(}\DecValTok{45}\NormalTok{)}
|
|
767
|
+
\FunctionTok{puts}\NormalTok{ exp10}
|
|
768
|
+
\end{Highlighting}
|
|
769
|
+
\end{Shaded}
|
|
770
|
+
|
|
771
|
+
\begin{verbatim}
|
|
772
|
+
## y <- 0.850903524534118
|
|
773
|
+
\end{verbatim}
|
|
774
|
+
|
|
775
|
+
\section{\texorpdfstring{Data manipulation wiht
|
|
776
|
+
\emph{dplyr}}{Data manipulation wiht dplyr}}\label{data-manipulation-wiht-dplyr}
|
|
777
|
+
|
|
778
|
+
In this section we will give a brief tour \emph{dplyr}'s usage in Galaaz
|
|
779
|
+
and how to manipulate data in Ruby with it. This section will follow
|
|
780
|
+
\href{https://dplyr.tidyverse.org/articles/dplyr.html}{\emph{dplyr}'s
|
|
781
|
+
vignette} that explores the nycflights13 data set. This dataset contains
|
|
782
|
+
all 336776 flights that departed from New York City in 2013. The data
|
|
783
|
+
comes from the US Bureau of Transportation Statistics.
|
|
784
|
+
|
|
785
|
+
Let's start by taking a look at this dataset:
|
|
786
|
+
|
|
787
|
+
\begin{Shaded}
|
|
788
|
+
\begin{Highlighting}[]
|
|
789
|
+
\ConstantTok{R}\AttributeTok{.library}\NormalTok{(}\VerbatimStringTok{\textquotesingle{}nycflights13\textquotesingle{}}\NormalTok{)}
|
|
790
|
+
\CommentTok{\# check it\textquotesingle{}s dimension}
|
|
791
|
+
\FunctionTok{puts} \OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:flights}\KeywordTok{]}\AttributeTok{.dim}
|
|
792
|
+
\CommentTok{\# and the structure}
|
|
793
|
+
\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:flights}\KeywordTok{]}\AttributeTok{.str}
|
|
794
|
+
\end{Highlighting}
|
|
795
|
+
\end{Shaded}
|
|
796
|
+
|
|
797
|
+
\begin{verbatim}
|
|
798
|
+
## ~(dim(flights))
|
|
799
|
+
## <environment: 0x57d8ad019298>
|
|
800
|
+
\end{verbatim}
|
|
801
|
+
|
|
802
|
+
Now, let's use a first verb of \emph{dplyr}: `filter'. This verb,
|
|
803
|
+
obviously, will filter the data by the given expression. In the next
|
|
804
|
+
block, we filter by columns `month' and `day'. The first argument to the
|
|
805
|
+
filter function is symbol `:flights'. A Ruby symbol, when given to an R
|
|
806
|
+
function will convert to the R variable of the same name, in this case
|
|
807
|
+
`flights', that holds the nycflights13 data frame.
|
|
808
|
+
|
|
809
|
+
The second and third arguments are expressions that will be used by the
|
|
810
|
+
filter function to filter by columns, looking for entries in which the
|
|
811
|
+
month and day are equal to 1.
|
|
812
|
+
|
|
813
|
+
\begin{Shaded}
|
|
814
|
+
\begin{Highlighting}[]
|
|
815
|
+
\FunctionTok{puts} \ConstantTok{R}\AttributeTok{.filter}\NormalTok{(}\WarningTok{:flights}\NormalTok{, (}\ConstantTok{R}\KeywordTok{[}\WarningTok{:month}\KeywordTok{]}\AttributeTok{.eq} \DecValTok{1}\NormalTok{), (}\ConstantTok{R}\KeywordTok{[}\WarningTok{:day}\KeywordTok{]}\AttributeTok{.eq} \DecValTok{1}\NormalTok{))}
|
|
816
|
+
\end{Highlighting}
|
|
817
|
+
\end{Shaded}
|
|
818
|
+
|
|
819
|
+
\begin{verbatim}
|
|
820
|
+
## # A tibble: 842 x 19
|
|
821
|
+
## year month day dep_time sched_dep_time dep_delay arr_time
|
|
822
|
+
## <int> <int> <int> <int> <int> <dbl> <int>
|
|
823
|
+
## 1 2013 1 1 517 515 2 830
|
|
824
|
+
## 2 2013 1 1 533 529 4 850
|
|
825
|
+
## 3 2013 1 1 542 540 2 923
|
|
826
|
+
## 4 2013 1 1 544 545 -1 1004
|
|
827
|
+
## 5 2013 1 1 554 600 -6 812
|
|
828
|
+
## 6 2013 1 1 554 558 -4 740
|
|
829
|
+
## 7 2013 1 1 555 600 -5 913
|
|
830
|
+
## 8 2013 1 1 557 600 -3 709
|
|
831
|
+
## 9 2013 1 1 557 600 -3 838
|
|
832
|
+
## 10 2013 1 1 558 600 -2 753
|
|
833
|
+
## # i 832 more rows
|
|
834
|
+
## # i 12 more variables: sched_arr_time <int>, arr_delay <dbl>,
|
|
835
|
+
## # carrier <chr>, flight <int>, tailnum <chr>, origin <chr>,
|
|
836
|
+
## # dest <chr>, air_time <dbl>, distance <dbl>, hour <dbl>,
|
|
837
|
+
## # minute <dbl>, time_hour <dttm>
|
|
838
|
+
\end{verbatim}
|
|
839
|
+
|
|
840
|
+
\subsection{\texorpdfstring{Programming with \emph{dplyr}: problems and
|
|
841
|
+
how to solve them in
|
|
842
|
+
Galaaz}{Programming with dplyr: problems and how to solve them in Galaaz}}\label{programming-with-dplyr-problems-and-how-to-solve-them-in-galaaz}
|
|
843
|
+
|
|
844
|
+
In this section we look at the list of problems that Hadley describes in
|
|
845
|
+
the ``Programming with dplyr'' vignette and show how those problems are
|
|
846
|
+
solved and coded with Galaaz. Readers interested in how those problems
|
|
847
|
+
are treated in \emph{dplyr} should read the vignette and use it as a
|
|
848
|
+
comparison with this blog.
|
|
849
|
+
|
|
850
|
+
\subsection{Filtering using
|
|
851
|
+
expressions}\label{filtering-using-expressions}
|
|
852
|
+
|
|
853
|
+
Now that we know how to write expressions and call R functions, let's do
|
|
854
|
+
some data manipulation in Galaaz. Let's first start by creating a data
|
|
855
|
+
frame. In R, the `data.frame' function creates a data frame. In Ruby,
|
|
856
|
+
writing `data.frame' will not parse as a single object. To call R
|
|
857
|
+
functions that have a `.' in them, we need to substitute the `.' with
|
|
858
|
+
'\_\_`. So, method 'data.frame' in R, is called in Galaaz as
|
|
859
|
+
`R.data\_\_frame':
|
|
860
|
+
|
|
861
|
+
\begin{Shaded}
|
|
862
|
+
\begin{Highlighting}[]
|
|
863
|
+
\NormalTok{df }\OperatorTok{=} \ConstantTok{R}\AttributeTok{.data\_\_frame}\NormalTok{(}\WarningTok{x:}\NormalTok{ (}\DecValTok{1}\OperatorTok{..}\DecValTok{3}\NormalTok{), }\WarningTok{y:}\NormalTok{ (}\DecValTok{3}\OperatorTok{..}\DecValTok{1}\NormalTok{))}
|
|
864
|
+
\FunctionTok{puts}\NormalTok{ df}
|
|
865
|
+
\end{Highlighting}
|
|
866
|
+
\end{Shaded}
|
|
867
|
+
|
|
868
|
+
\begin{verbatim}
|
|
869
|
+
## x y
|
|
870
|
+
## 1 1 3
|
|
871
|
+
## 2 2 2
|
|
872
|
+
## 3 3 1
|
|
873
|
+
\end{verbatim}
|
|
874
|
+
|
|
875
|
+
\emph{dplyr} provides the `filter' function, that filters data in a data
|
|
876
|
+
brame. The `filter' function can be called on this data frame either by
|
|
877
|
+
using `R.filter(df, \ldots)' or by using dot notation.
|
|
878
|
+
|
|
879
|
+
-------FIX---------
|
|
880
|
+
|
|
881
|
+
We prefer to use dot notation as shown below. The argument to `filter'
|
|
882
|
+
should be an expression. Note that if we gave to filter a Ruby
|
|
883
|
+
expression such as `x == 1', we would get an error, since there is no
|
|
884
|
+
variable `x' defined and if `x' was a variable then `x == 1' would
|
|
885
|
+
either be `true' or `false'. Our goal is to filter our data frame
|
|
886
|
+
returning all rows in which the `x' value is equal to 1. To express this
|
|
887
|
+
we want: `R{[}:x{]}.eq 1', where :x will be interpreted by filter as the
|
|
888
|
+
`x' column.
|
|
889
|
+
|
|
890
|
+
\begin{Shaded}
|
|
891
|
+
\begin{Highlighting}[]
|
|
892
|
+
\FunctionTok{puts}\NormalTok{ df}\AttributeTok{.filter}\NormalTok{(}\ConstantTok{R}\KeywordTok{[}\WarningTok{:x}\KeywordTok{]}\AttributeTok{.eq} \DecValTok{1}\NormalTok{)}
|
|
893
|
+
\end{Highlighting}
|
|
894
|
+
\end{Shaded}
|
|
895
|
+
|
|
896
|
+
\begin{verbatim}
|
|
897
|
+
## x y
|
|
898
|
+
## 1 1 3
|
|
899
|
+
\end{verbatim}
|
|
900
|
+
|
|
901
|
+
In R, and when coding with `tidyverse', arguments to a function are
|
|
902
|
+
usually not \emph{referencially transparent}. That is, you can't replace
|
|
903
|
+
a value with a seemingly equivalent object that you've defined
|
|
904
|
+
elsewhere. In other words, this code
|
|
905
|
+
|
|
906
|
+
\begin{Shaded}
|
|
907
|
+
\begin{Highlighting}[]
|
|
908
|
+
\NormalTok{my\_var }\OtherTok{\textless{}{-}}\NormalTok{ x}
|
|
909
|
+
\FunctionTok{filter}\NormalTok{(df, my\_var }\SpecialCharTok{==} \DecValTok{1}\NormalTok{)}
|
|
910
|
+
\end{Highlighting}
|
|
911
|
+
\end{Shaded}
|
|
912
|
+
|
|
913
|
+
Generates the following error: ``object `x' not found.
|
|
914
|
+
|
|
915
|
+
However, in Galaaz, arguments are referencially transparent as can be
|
|
916
|
+
seen by the code below. Note initially that `my\_var = R{[}:x{]}' will
|
|
917
|
+
not give the error ``object `x' not found'' since `:x' is treated as an
|
|
918
|
+
expression and assigned to my\_var. Then when doing (my\_var.eq 1),
|
|
919
|
+
my\_var is a variable that resolves to `:x' and it becomes equivalent to
|
|
920
|
+
(R{[}:x{]}.eq 1) which is what we want.
|
|
921
|
+
|
|
922
|
+
\begin{Shaded}
|
|
923
|
+
\begin{Highlighting}[]
|
|
924
|
+
\NormalTok{my\_var }\OperatorTok{=} \ConstantTok{R}\KeywordTok{[}\WarningTok{:x}\KeywordTok{]}
|
|
925
|
+
\FunctionTok{puts}\NormalTok{ df}\AttributeTok{.filter}\NormalTok{(my\_var}\AttributeTok{.eq} \DecValTok{1}\NormalTok{)}
|
|
926
|
+
\end{Highlighting}
|
|
927
|
+
\end{Shaded}
|
|
928
|
+
|
|
929
|
+
\begin{verbatim}
|
|
930
|
+
## x y
|
|
931
|
+
## 1 1 3
|
|
932
|
+
\end{verbatim}
|
|
933
|
+
|
|
934
|
+
As stated by Hadley
|
|
935
|
+
|
|
936
|
+
\begin{quote}
|
|
937
|
+
dplyr code is ambiguous. Depending on what variables are defined where,
|
|
938
|
+
filter(df, x == y) could be equivalent to any of:
|
|
939
|
+
\end{quote}
|
|
940
|
+
|
|
941
|
+
\begin{verbatim}
|
|
942
|
+
df[df$x == df$y, ]
|
|
943
|
+
df[df$x == y, ]
|
|
944
|
+
df[x == df$y, ]
|
|
945
|
+
df[x == y, ]
|
|
946
|
+
\end{verbatim}
|
|
947
|
+
|
|
948
|
+
In galaaz this ambiguity does not exist, filter(df, x.eq y) is not a
|
|
949
|
+
valid expression as expressions are build with symbols. In doing
|
|
950
|
+
filter(df, R{[}:x{]}.eq y) we are looking for elements of the `x' column
|
|
951
|
+
that are equal to a previously defined y variable. Finally in filter(df,
|
|
952
|
+
R{[}:x{]}.eq R{[}:y{]}) we are looking for elements in which the `x'
|
|
953
|
+
column value is equal to the `y' column value. This can be seen in the
|
|
954
|
+
following two chunks of code:
|
|
955
|
+
|
|
956
|
+
\begin{Shaded}
|
|
957
|
+
\begin{Highlighting}[]
|
|
958
|
+
\NormalTok{y }\OperatorTok{=} \DecValTok{1}
|
|
959
|
+
\NormalTok{x }\OperatorTok{=} \DecValTok{2}
|
|
960
|
+
|
|
961
|
+
\CommentTok{\# looking for values where the \textquotesingle{}x\textquotesingle{} column is equal to the \textquotesingle{}y\textquotesingle{} column}
|
|
962
|
+
\FunctionTok{puts}\NormalTok{ df}\AttributeTok{.filter}\NormalTok{(}\ConstantTok{R}\KeywordTok{[}\WarningTok{:x}\KeywordTok{]}\AttributeTok{.eq} \ConstantTok{R}\KeywordTok{[}\WarningTok{:y}\KeywordTok{]}\NormalTok{)}
|
|
963
|
+
\end{Highlighting}
|
|
964
|
+
\end{Shaded}
|
|
965
|
+
|
|
966
|
+
\begin{verbatim}
|
|
967
|
+
## x y
|
|
968
|
+
## 1 2 2
|
|
969
|
+
\end{verbatim}
|
|
970
|
+
|
|
971
|
+
\begin{Shaded}
|
|
972
|
+
\begin{Highlighting}[]
|
|
973
|
+
\CommentTok{\# looking for values where the \textquotesingle{}x\textquotesingle{} column is equal to the \textquotesingle{}y\textquotesingle{} variable}
|
|
974
|
+
\CommentTok{\# in this case, the number 1}
|
|
975
|
+
\FunctionTok{puts}\NormalTok{ df}\AttributeTok{.filter}\NormalTok{(}\ConstantTok{R}\KeywordTok{[}\WarningTok{:x}\KeywordTok{]}\AttributeTok{.eq}\NormalTok{ y)}
|
|
976
|
+
\end{Highlighting}
|
|
977
|
+
\end{Shaded}
|
|
978
|
+
|
|
979
|
+
\begin{verbatim}
|
|
980
|
+
## x y
|
|
981
|
+
## 1 1 3
|
|
982
|
+
\end{verbatim}
|
|
983
|
+
|
|
984
|
+
\subsection{Writing a function that applies to different data
|
|
985
|
+
sets}\label{writing-a-function-that-applies-to-different-data-sets}
|
|
986
|
+
|
|
987
|
+
Let's suppose that we want to write a function that receives as the
|
|
988
|
+
first argument a data frame and as second argument an expression that
|
|
989
|
+
adds a column to the data frame that is equal to the sum of elements in
|
|
990
|
+
column `a' plus `x'.
|
|
991
|
+
|
|
992
|
+
Here is the intended behaviour using the `mutate' function of `dplyr':
|
|
993
|
+
|
|
994
|
+
\begin{verbatim}
|
|
995
|
+
mutate(df1, y = a + x)
|
|
996
|
+
mutate(df2, y = a + x)
|
|
997
|
+
mutate(df3, y = a + x)
|
|
998
|
+
mutate(df4, y = a + x)
|
|
999
|
+
\end{verbatim}
|
|
1000
|
+
|
|
1001
|
+
The naive approach to writing an R function to solve this problem is:
|
|
1002
|
+
|
|
1003
|
+
\begin{verbatim}
|
|
1004
|
+
mutate_y <- function(df) {
|
|
1005
|
+
mutate(df, y = a + x)
|
|
1006
|
+
}
|
|
1007
|
+
\end{verbatim}
|
|
1008
|
+
|
|
1009
|
+
Unfortunately, in R, this function can fail silently if one of the
|
|
1010
|
+
variables isn't present in the data frame, but is present in the global
|
|
1011
|
+
environment. We will not go through here how to solve this problem in R.
|
|
1012
|
+
|
|
1013
|
+
In Galaaz the method mutate\_y below will work fine and will never fail
|
|
1014
|
+
silently.
|
|
1015
|
+
|
|
1016
|
+
\begin{Shaded}
|
|
1017
|
+
\begin{Highlighting}[]
|
|
1018
|
+
\ControlFlowTok{def}\NormalTok{ mutate\_y(df)}
|
|
1019
|
+
\CommentTok{\# Column names are Ruby kwargs (y: …).}
|
|
1020
|
+
\CommentTok{\# Use .assign only for R \textasciigrave{}\textless{}{-}\textasciigrave{} expressions.}
|
|
1021
|
+
\NormalTok{ df}\AttributeTok{.mutate}\NormalTok{(}\WarningTok{y:} \ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{+} \ConstantTok{R}\KeywordTok{[}\WarningTok{:x}\KeywordTok{]}\NormalTok{)}
|
|
1022
|
+
\ControlFlowTok{end}
|
|
1023
|
+
\end{Highlighting}
|
|
1024
|
+
\end{Shaded}
|
|
1025
|
+
|
|
1026
|
+
Here we create a data frame that has only one column named `x':
|
|
1027
|
+
|
|
1028
|
+
\begin{Shaded}
|
|
1029
|
+
\begin{Highlighting}[]
|
|
1030
|
+
\NormalTok{df1 }\OperatorTok{=} \ConstantTok{R}\AttributeTok{.data\_\_frame}\NormalTok{(}\WarningTok{x:}\NormalTok{ (}\DecValTok{1}\OperatorTok{..}\DecValTok{3}\NormalTok{))}
|
|
1031
|
+
\FunctionTok{puts}\NormalTok{ df1}
|
|
1032
|
+
\end{Highlighting}
|
|
1033
|
+
\end{Shaded}
|
|
1034
|
+
|
|
1035
|
+
\begin{verbatim}
|
|
1036
|
+
## x
|
|
1037
|
+
## 1 1
|
|
1038
|
+
## 2 2
|
|
1039
|
+
## 3 3
|
|
1040
|
+
\end{verbatim}
|
|
1041
|
+
|
|
1042
|
+
Note that method mutate\_y will fail independetly from the fact that
|
|
1043
|
+
variable `a' is defined and in the scope of the method. Variable `a' has
|
|
1044
|
+
no relationship with the symbol \texttt{R{[}:a{]}} used in the
|
|
1045
|
+
definition of `mutate\_y' above:
|
|
1046
|
+
|
|
1047
|
+
\begin{Shaded}
|
|
1048
|
+
\begin{Highlighting}[]
|
|
1049
|
+
\NormalTok{a }\OperatorTok{=} \DecValTok{10}
|
|
1050
|
+
\ControlFlowTok{begin}
|
|
1051
|
+
\NormalTok{ mutate\_y(df1)}
|
|
1052
|
+
\ControlFlowTok{rescue} \OperatorTok{=\textgreater{}}\NormalTok{ e}
|
|
1053
|
+
\CommentTok{\# Short message only — full backtraces overflow PDF code boxes.}
|
|
1054
|
+
\FunctionTok{puts} \StringTok{"}\SpecialCharTok{\#\{}\NormalTok{e}\AttributeTok{.class}\SpecialCharTok{\}}\StringTok{: }\SpecialCharTok{\#\{}\NormalTok{e}\AttributeTok{.message}\SpecialCharTok{\}}\StringTok{"}
|
|
1055
|
+
\ControlFlowTok{end}
|
|
1056
|
+
\end{Highlighting}
|
|
1057
|
+
\end{Shaded}
|
|
1058
|
+
|
|
1059
|
+
\begin{verbatim}
|
|
1060
|
+
## NewBridge::SessionClient::RProcessError: Error: i In argument: `y = a + x`.
|
|
1061
|
+
## Caused by error:
|
|
1062
|
+
## ! object 'a' not found
|
|
1063
|
+
\end{verbatim}
|
|
1064
|
+
|
|
1065
|
+
\subsection{Different expressions}\label{different-expressions}
|
|
1066
|
+
|
|
1067
|
+
Let's move to the next problem as presented by Hadley where trying to
|
|
1068
|
+
write a function in R that will receive two argumens, the first a
|
|
1069
|
+
variable and the second an expression is not trivial. Below we create a
|
|
1070
|
+
data frame and we want to write a function that groups data by a
|
|
1071
|
+
variable and summarises it by an expression:
|
|
1072
|
+
|
|
1073
|
+
\begin{Shaded}
|
|
1074
|
+
\begin{Highlighting}[]
|
|
1075
|
+
\FunctionTok{set.seed}\NormalTok{(}\DecValTok{123}\NormalTok{)}
|
|
1076
|
+
|
|
1077
|
+
\NormalTok{df }\OtherTok{\textless{}{-}} \FunctionTok{data.frame}\NormalTok{(}
|
|
1078
|
+
\AttributeTok{g1 =} \FunctionTok{c}\NormalTok{(}\DecValTok{1}\NormalTok{, }\DecValTok{1}\NormalTok{, }\DecValTok{2}\NormalTok{, }\DecValTok{2}\NormalTok{, }\DecValTok{2}\NormalTok{),}
|
|
1079
|
+
\AttributeTok{g2 =} \FunctionTok{c}\NormalTok{(}\DecValTok{1}\NormalTok{, }\DecValTok{2}\NormalTok{, }\DecValTok{1}\NormalTok{, }\DecValTok{2}\NormalTok{, }\DecValTok{1}\NormalTok{),}
|
|
1080
|
+
\AttributeTok{a =} \FunctionTok{sample}\NormalTok{(}\DecValTok{5}\NormalTok{),}
|
|
1081
|
+
\AttributeTok{b =} \FunctionTok{sample}\NormalTok{(}\DecValTok{5}\NormalTok{)}
|
|
1082
|
+
\NormalTok{)}
|
|
1083
|
+
|
|
1084
|
+
\FunctionTok{as.data.frame}\NormalTok{(df)}
|
|
1085
|
+
\end{Highlighting}
|
|
1086
|
+
\end{Shaded}
|
|
1087
|
+
|
|
1088
|
+
\begin{verbatim}
|
|
1089
|
+
## g1 g2 a b
|
|
1090
|
+
## 1 1 1 3 3
|
|
1091
|
+
## 2 1 2 2 1
|
|
1092
|
+
## 3 2 1 5 2
|
|
1093
|
+
## 4 2 2 4 5
|
|
1094
|
+
## 5 2 1 1 4
|
|
1095
|
+
\end{verbatim}
|
|
1096
|
+
|
|
1097
|
+
\begin{Shaded}
|
|
1098
|
+
\begin{Highlighting}[]
|
|
1099
|
+
\NormalTok{d2 }\OtherTok{\textless{}{-}}\NormalTok{ df }\SpecialCharTok{\%\textgreater{}\%}
|
|
1100
|
+
\FunctionTok{group\_by}\NormalTok{(g1) }\SpecialCharTok{\%\textgreater{}\%}
|
|
1101
|
+
\FunctionTok{summarise}\NormalTok{(}\AttributeTok{a =} \FunctionTok{mean}\NormalTok{(a))}
|
|
1102
|
+
|
|
1103
|
+
\FunctionTok{as.data.frame}\NormalTok{(d2)}
|
|
1104
|
+
\end{Highlighting}
|
|
1105
|
+
\end{Shaded}
|
|
1106
|
+
|
|
1107
|
+
\begin{verbatim}
|
|
1108
|
+
## g1 a
|
|
1109
|
+
## 1 1 2.500000
|
|
1110
|
+
## 2 2 3.333333
|
|
1111
|
+
\end{verbatim}
|
|
1112
|
+
|
|
1113
|
+
\begin{Shaded}
|
|
1114
|
+
\begin{Highlighting}[]
|
|
1115
|
+
\NormalTok{d2 }\OtherTok{\textless{}{-}}\NormalTok{ df }\SpecialCharTok{\%\textgreater{}\%}
|
|
1116
|
+
\FunctionTok{group\_by}\NormalTok{(g2) }\SpecialCharTok{\%\textgreater{}\%}
|
|
1117
|
+
\FunctionTok{summarise}\NormalTok{(}\AttributeTok{a =} \FunctionTok{mean}\NormalTok{(a))}
|
|
1118
|
+
|
|
1119
|
+
\FunctionTok{as.data.frame}\NormalTok{(d2) }
|
|
1120
|
+
\end{Highlighting}
|
|
1121
|
+
\end{Shaded}
|
|
1122
|
+
|
|
1123
|
+
\begin{verbatim}
|
|
1124
|
+
## g2 a
|
|
1125
|
+
## 1 1 3
|
|
1126
|
+
## 2 2 3
|
|
1127
|
+
\end{verbatim}
|
|
1128
|
+
|
|
1129
|
+
As shown by Hadley, one might expect this function to do the trick:
|
|
1130
|
+
|
|
1131
|
+
\begin{Shaded}
|
|
1132
|
+
\begin{Highlighting}[]
|
|
1133
|
+
\NormalTok{my\_summarise }\OtherTok{\textless{}{-}} \ControlFlowTok{function}\NormalTok{(df, group\_var) \{}
|
|
1134
|
+
\NormalTok{ df }\SpecialCharTok{\%\textgreater{}\%}
|
|
1135
|
+
\FunctionTok{group\_by}\NormalTok{(group\_var) }\SpecialCharTok{\%\textgreater{}\%}
|
|
1136
|
+
\FunctionTok{summarise}\NormalTok{(}\AttributeTok{a =} \FunctionTok{mean}\NormalTok{(a))}
|
|
1137
|
+
\NormalTok{\}}
|
|
1138
|
+
|
|
1139
|
+
\CommentTok{\# my\_summarise(df, g1)}
|
|
1140
|
+
\CommentTok{\#\textgreater{} Error: Column \textasciigrave{}group\_var\textasciigrave{} is unknown}
|
|
1141
|
+
\end{Highlighting}
|
|
1142
|
+
\end{Shaded}
|
|
1143
|
+
|
|
1144
|
+
In order to solve this problem, coding with dplyr requires the
|
|
1145
|
+
introduction of many new concepts and functions such as `quo', `quos',
|
|
1146
|
+
`enquo', `enquos', `!!' (bang bang), `!!!' (triple bang). Again, we'll
|
|
1147
|
+
leave to Hadley the explanation on how to use all those functions.
|
|
1148
|
+
|
|
1149
|
+
Now, let's try to implement the same function in galaaz. The next code
|
|
1150
|
+
block first prints the `df' data frame defined previously in R (to
|
|
1151
|
+
access an R variable from Galaaz, we use the tilde operator
|
|
1152
|
+
`\textasciitilde{}' applied to the R variable name as symbol, i.e.,
|
|
1153
|
+
`:df'. We then create the `my\_summarize' method and call it passing the
|
|
1154
|
+
R data frame and the group by variable `:g1':
|
|
1155
|
+
|
|
1156
|
+
\begin{Shaded}
|
|
1157
|
+
\begin{Highlighting}[]
|
|
1158
|
+
\FunctionTok{puts} \OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:df}\KeywordTok{]}
|
|
1159
|
+
\FunctionTok{print} \StringTok{"}
|
|
1160
|
+
\StringTok{"}
|
|
1161
|
+
|
|
1162
|
+
\ControlFlowTok{def}\NormalTok{ my\_summarize(df, group\_var)}
|
|
1163
|
+
\NormalTok{ df}\AttributeTok{.group\_by}\NormalTok{(group\_var)}\AttributeTok{.}
|
|
1164
|
+
\AttributeTok{summarize}\NormalTok{(}\WarningTok{a:} \ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]}\AttributeTok{.mean}\NormalTok{)}
|
|
1165
|
+
\ControlFlowTok{end}
|
|
1166
|
+
|
|
1167
|
+
\FunctionTok{puts}\NormalTok{ my\_summarize(}\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:df}\KeywordTok{]}\NormalTok{, }\ConstantTok{R}\KeywordTok{[}\WarningTok{:g1}\KeywordTok{]}\NormalTok{)}
|
|
1168
|
+
\end{Highlighting}
|
|
1169
|
+
\end{Shaded}
|
|
1170
|
+
|
|
1171
|
+
\begin{verbatim}
|
|
1172
|
+
## g1 g2 a b
|
|
1173
|
+
## 1 1 1 3 3
|
|
1174
|
+
## 2 1 2 2 1
|
|
1175
|
+
## 3 2 1 5 2
|
|
1176
|
+
## 4 2 2 4 5
|
|
1177
|
+
## 5 2 1 1 4
|
|
1178
|
+
##
|
|
1179
|
+
## # A tibble: 2 x 2
|
|
1180
|
+
## g1 a
|
|
1181
|
+
## <dbl> <dbl>
|
|
1182
|
+
## 1 1 2.5
|
|
1183
|
+
## 2 2 3.33
|
|
1184
|
+
\end{verbatim}
|
|
1185
|
+
|
|
1186
|
+
It works!!! Well, let's make sure this was not just some coincidence
|
|
1187
|
+
|
|
1188
|
+
\begin{Shaded}
|
|
1189
|
+
\begin{Highlighting}[]
|
|
1190
|
+
\FunctionTok{puts}\NormalTok{ my\_summarize(}\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:df}\KeywordTok{]}\NormalTok{, }\ConstantTok{R}\KeywordTok{[}\WarningTok{:g2}\KeywordTok{]}\NormalTok{)}
|
|
1191
|
+
\end{Highlighting}
|
|
1192
|
+
\end{Shaded}
|
|
1193
|
+
|
|
1194
|
+
\begin{verbatim}
|
|
1195
|
+
## # A tibble: 2 x 2
|
|
1196
|
+
## g2 a
|
|
1197
|
+
## <dbl> <dbl>
|
|
1198
|
+
## 1 1 3
|
|
1199
|
+
## 2 2 3
|
|
1200
|
+
\end{verbatim}
|
|
1201
|
+
|
|
1202
|
+
Great, everything is fine! No magic, no new functions, no complexities,
|
|
1203
|
+
just normal, standard Ruby code. If you've ever done NSE in R, this
|
|
1204
|
+
certainly feels much safer and easy to implement.
|
|
1205
|
+
|
|
1206
|
+
\subsection{Different input variables}\label{different-input-variables}
|
|
1207
|
+
|
|
1208
|
+
In the previous section we've managed to get rid of all NSE formulation
|
|
1209
|
+
for a simple example, but does this remain true for more complex
|
|
1210
|
+
examples, or will the Galaaz way prove inpractical for more complex
|
|
1211
|
+
code?
|
|
1212
|
+
|
|
1213
|
+
In the next example Hadley proposes us to write a function that given an
|
|
1214
|
+
expression such as `a' or `a * b', calculates three summaries. What we
|
|
1215
|
+
want a function that does the same as these R statements:
|
|
1216
|
+
|
|
1217
|
+
\begin{verbatim}
|
|
1218
|
+
summarise(df, mean = mean(a), sum = sum(a), n = n())
|
|
1219
|
+
#> # A tibble: 1 x 3
|
|
1220
|
+
#> mean sum n
|
|
1221
|
+
#> <dbl> <int> <int>
|
|
1222
|
+
#> 1 3 15 5
|
|
1223
|
+
|
|
1224
|
+
summarise(df, mean = mean(a * b), sum = sum(a * b), n = n())
|
|
1225
|
+
#> # A tibble: 1 x 3
|
|
1226
|
+
#> mean sum n
|
|
1227
|
+
#> <dbl> <int> <int>
|
|
1228
|
+
#> 1 9 45 5
|
|
1229
|
+
\end{verbatim}
|
|
1230
|
+
|
|
1231
|
+
Let's try it in galaaz:
|
|
1232
|
+
|
|
1233
|
+
\begin{Shaded}
|
|
1234
|
+
\begin{Highlighting}[]
|
|
1235
|
+
\ControlFlowTok{def}\NormalTok{ my\_summarise2(df, expr)}
|
|
1236
|
+
\NormalTok{ df}\AttributeTok{.summarize}\NormalTok{(}
|
|
1237
|
+
\WarningTok{mean:} \ConstantTok{E}\AttributeTok{.mean}\NormalTok{(expr),}
|
|
1238
|
+
\WarningTok{sum:} \ConstantTok{E}\AttributeTok{.sum}\NormalTok{(expr),}
|
|
1239
|
+
\WarningTok{n:} \ConstantTok{E}\AttributeTok{.n}
|
|
1240
|
+
\NormalTok{ )}
|
|
1241
|
+
\ControlFlowTok{end}
|
|
1242
|
+
|
|
1243
|
+
\FunctionTok{puts}\NormalTok{ my\_summarise2((}\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:df}\KeywordTok{]}\NormalTok{), }\WarningTok{:a}\NormalTok{)}
|
|
1244
|
+
\FunctionTok{puts}\NormalTok{ my\_summarise2((}\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:df}\KeywordTok{]}\NormalTok{), }\ConstantTok{R}\KeywordTok{[}\WarningTok{:a}\KeywordTok{]} \OperatorTok{*} \ConstantTok{R}\KeywordTok{[}\WarningTok{:b}\KeywordTok{]}\NormalTok{)}
|
|
1245
|
+
\end{Highlighting}
|
|
1246
|
+
\end{Shaded}
|
|
1247
|
+
|
|
1248
|
+
\begin{verbatim}
|
|
1249
|
+
## mean sum n
|
|
1250
|
+
## 1 3 15 5
|
|
1251
|
+
## mean sum n
|
|
1252
|
+
## 1 9 45 5
|
|
1253
|
+
\end{verbatim}
|
|
1254
|
+
|
|
1255
|
+
Once again, there is no need to use any special theory or functions. The
|
|
1256
|
+
only point to be careful about is the use of `E' to build expressions
|
|
1257
|
+
from functions `mean', `sum' and `n'.
|
|
1258
|
+
|
|
1259
|
+
\subsection{Different input and output
|
|
1260
|
+
variable}\label{different-input-and-output-variable}
|
|
1261
|
+
|
|
1262
|
+
Now the next challenge presented by Hadley is to vary the name of the
|
|
1263
|
+
output variables based on the received expression. So, if the input
|
|
1264
|
+
expression is `a', we want our data frame columns to be named `mean\_a'
|
|
1265
|
+
and `sum\_a'. Now, if the input expression is `b', columns should be
|
|
1266
|
+
named `mean\_b' and `sum\_b'.
|
|
1267
|
+
|
|
1268
|
+
\begin{verbatim}
|
|
1269
|
+
mutate(df, mean_a = mean(a), sum_a = sum(a))
|
|
1270
|
+
#> # A tibble: 5 x 6
|
|
1271
|
+
#> g1 g2 a b mean_a sum_a
|
|
1272
|
+
#> <dbl> <dbl> <int> <int> <dbl> <int>
|
|
1273
|
+
#> 1 1 1 1 3 3 15
|
|
1274
|
+
#> 2 1 2 4 2 3 15
|
|
1275
|
+
#> 3 2 1 2 1 3 15
|
|
1276
|
+
#> 4 2 2 5 4 3 15
|
|
1277
|
+
#> # … with 1 more row
|
|
1278
|
+
|
|
1279
|
+
mutate(df, mean_b = mean(b), sum_b = sum(b))
|
|
1280
|
+
#> # A tibble: 5 x 6
|
|
1281
|
+
#> g1 g2 a b mean_b sum_b
|
|
1282
|
+
#> <dbl> <dbl> <int> <int> <dbl> <int>
|
|
1283
|
+
#> 1 1 1 1 3 3 15
|
|
1284
|
+
#> 2 1 2 4 2 3 15
|
|
1285
|
+
#> 3 2 1 2 1 3 15
|
|
1286
|
+
#> 4 2 2 5 4 3 15
|
|
1287
|
+
#> # … with 1 more row
|
|
1288
|
+
\end{verbatim}
|
|
1289
|
+
|
|
1290
|
+
In order to solve this problem in R, Hadley needs to introduce some more
|
|
1291
|
+
new functions and notations: `quo\_name' and the `:=' operator from
|
|
1292
|
+
package `rlang'
|
|
1293
|
+
|
|
1294
|
+
Here is our Ruby code:
|
|
1295
|
+
|
|
1296
|
+
\begin{Shaded}
|
|
1297
|
+
\begin{Highlighting}[]
|
|
1298
|
+
\ControlFlowTok{def}\NormalTok{ my\_mutate(df, expr)}
|
|
1299
|
+
\NormalTok{ mean\_name }\OperatorTok{=} \StringTok{"mean\_}\SpecialCharTok{\#\{}\NormalTok{expr}\AttributeTok{.to\_s}\SpecialCharTok{\}}\StringTok{"}
|
|
1300
|
+
\NormalTok{ sum\_name }\OperatorTok{=} \StringTok{"sum\_}\SpecialCharTok{\#\{}\NormalTok{expr}\AttributeTok{.to\_s}\SpecialCharTok{\}}\StringTok{"}
|
|
1301
|
+
|
|
1302
|
+
\NormalTok{ df}\AttributeTok{.mutate}\NormalTok{(mean\_name }\OperatorTok{=\textgreater{}} \ConstantTok{E}\AttributeTok{.mean}\NormalTok{(expr),}
|
|
1303
|
+
\NormalTok{ sum\_name }\OperatorTok{=\textgreater{}} \ConstantTok{E}\AttributeTok{.sum}\NormalTok{(expr))}
|
|
1304
|
+
\ControlFlowTok{end}
|
|
1305
|
+
|
|
1306
|
+
\FunctionTok{puts}\NormalTok{ my\_mutate((}\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:df}\KeywordTok{]}\NormalTok{), }\WarningTok{:a}\NormalTok{)}
|
|
1307
|
+
\FunctionTok{puts}\NormalTok{ my\_mutate((}\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:df}\KeywordTok{]}\NormalTok{), }\WarningTok{:b}\NormalTok{)}
|
|
1308
|
+
\end{Highlighting}
|
|
1309
|
+
\end{Shaded}
|
|
1310
|
+
|
|
1311
|
+
\begin{verbatim}
|
|
1312
|
+
## g1 g2 a b mean_a sum_a
|
|
1313
|
+
## 1 1 1 3 3 3 15
|
|
1314
|
+
## 2 1 2 2 1 3 15
|
|
1315
|
+
## 3 2 1 5 2 3 15
|
|
1316
|
+
## 4 2 2 4 5 3 15
|
|
1317
|
+
## 5 2 1 1 4 3 15
|
|
1318
|
+
## g1 g2 a b mean_b sum_b
|
|
1319
|
+
## 1 1 1 3 3 3 15
|
|
1320
|
+
## 2 1 2 2 1 3 15
|
|
1321
|
+
## 3 2 1 5 2 3 15
|
|
1322
|
+
## 4 2 2 4 5 3 15
|
|
1323
|
+
## 5 2 1 1 4 3 15
|
|
1324
|
+
\end{verbatim}
|
|
1325
|
+
|
|
1326
|
+
It really seems that ``Non Standard Evaluation'' is actually quite
|
|
1327
|
+
standard in Galaaz! But, you might have noticed a small change in the
|
|
1328
|
+
way the arguments to the mutate method were called. In a previous
|
|
1329
|
+
example we used df.summarise(mean: E.mean(:a), \ldots) where the column
|
|
1330
|
+
name was followed by a `:' colom. In this example, we have
|
|
1331
|
+
df.mutate(mean\_name =\textgreater{} E.mean(expr), \ldots) and variable
|
|
1332
|
+
mean\_name is not followed by `:' but by `=\textgreater{}'. This is
|
|
1333
|
+
standard Ruby notation.
|
|
1334
|
+
|
|
1335
|
+
{[}explain\ldots.{]}
|
|
1336
|
+
|
|
1337
|
+
\subsection{Capturing multiple
|
|
1338
|
+
variables}\label{capturing-multiple-variables}
|
|
1339
|
+
|
|
1340
|
+
Moving on with new complexities, Hadley proposes us to solve the problem
|
|
1341
|
+
in which the summarise function will receive any number of grouping
|
|
1342
|
+
variables.
|
|
1343
|
+
|
|
1344
|
+
This again is quite standard Ruby. In order to receive an undefined
|
|
1345
|
+
number of paramenters the paramenter is preceded by '*':
|
|
1346
|
+
|
|
1347
|
+
\begin{Shaded}
|
|
1348
|
+
\begin{Highlighting}[]
|
|
1349
|
+
\ControlFlowTok{def}\NormalTok{ my\_summarise3(df, }\OperatorTok{*}\NormalTok{group\_vars)}
|
|
1350
|
+
\NormalTok{ df}\AttributeTok{.group\_by}\NormalTok{(}\OperatorTok{*}\NormalTok{group\_vars)}\AttributeTok{.}
|
|
1351
|
+
\AttributeTok{summarise}\NormalTok{(}\WarningTok{a:} \ConstantTok{E}\AttributeTok{.mean}\NormalTok{(}\WarningTok{:a}\NormalTok{))}
|
|
1352
|
+
\ControlFlowTok{end}
|
|
1353
|
+
|
|
1354
|
+
\FunctionTok{puts}\NormalTok{ my\_summarise3((}\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:df}\KeywordTok{]}\NormalTok{), }\WarningTok{:g1}\NormalTok{, }\WarningTok{:g2}\NormalTok{)}
|
|
1355
|
+
\end{Highlighting}
|
|
1356
|
+
\end{Shaded}
|
|
1357
|
+
|
|
1358
|
+
\begin{verbatim}
|
|
1359
|
+
## # A tibble: 4 x 3
|
|
1360
|
+
## # Groups: g1 [2]
|
|
1361
|
+
## g1 g2 a
|
|
1362
|
+
## <dbl> <dbl> <dbl>
|
|
1363
|
+
## 1 1 1 3
|
|
1364
|
+
## 2 1 2 2
|
|
1365
|
+
## 3 2 1 3
|
|
1366
|
+
## 4 2 2 4
|
|
1367
|
+
\end{verbatim}
|
|
1368
|
+
|
|
1369
|
+
\section{Why does R require NSE and Galaaz does
|
|
1370
|
+
not?}\label{why-does-r-require-nse-and-galaaz-does-not}
|
|
1371
|
+
|
|
1372
|
+
NSE introduces a number of new concepts, such as `quoting',
|
|
1373
|
+
`quasiquotation', `unquoting' and `unquote-splicing', while in Galaaz
|
|
1374
|
+
none of those concepts are needed. What gives?
|
|
1375
|
+
|
|
1376
|
+
R is an extremely flexible language and it has lazy evaluation of
|
|
1377
|
+
parameters. When in R a function is called as `summarise(df, a = b)',
|
|
1378
|
+
the summarise function receives the litteral `a = b' parameter and can
|
|
1379
|
+
work with this as if it were a string. In R, it is not clear what a and
|
|
1380
|
+
b are, they can be expressions or they can be variables, it is up to the
|
|
1381
|
+
function to decide what `a = b' means.
|
|
1382
|
+
|
|
1383
|
+
In Ruby, there is no lazy evaluation of parameters and `a' is always a
|
|
1384
|
+
variable and so is `b'. Variables assume their value as soon as they are
|
|
1385
|
+
used, so `x = a' is immediately evaluate and variable `x' will receive
|
|
1386
|
+
the value of variable `a' as soon as the Ruby statement is executed.
|
|
1387
|
+
Ruby also provides the notion of a symbol; `:a' is a symbol and does not
|
|
1388
|
+
evaluate to anything. Galaaz uses Ruby symbols to build expressions that
|
|
1389
|
+
are not bound to anything: `R{[}:a{]}.eq R{[}:b{]}' is clearly an
|
|
1390
|
+
expression and has no relationship whatsoever with the statment `a = b'.
|
|
1391
|
+
By using symbols, variables and expressions all the possible ambiguities
|
|
1392
|
+
that are found in R are eliminated in Galaaz.
|
|
1393
|
+
|
|
1394
|
+
The main problem that remains, is that in R, functions are not clearly
|
|
1395
|
+
documented as what type of input they are expecting, they might be
|
|
1396
|
+
expecting regular variables or they might be expecting expressions and
|
|
1397
|
+
the R function will know how to deal with an input of the form `a = b',
|
|
1398
|
+
now for the Ruby developer it might not be immediately clear if it
|
|
1399
|
+
should call the function passing the value `true' if variable `a' is
|
|
1400
|
+
equal to variable `b' or if it should call the function passing the
|
|
1401
|
+
expression `R{[}:a{]}.eq R{[}:b{]}'.
|
|
1402
|
+
|
|
1403
|
+
\section{Advanced dplyr features}\label{advanced-dplyr-features}
|
|
1404
|
+
|
|
1405
|
+
In the blog:
|
|
1406
|
+
\href{https://www.r-bloggers.com/programming-with-dplyr-by-using-dplyr/}{Programming
|
|
1407
|
+
with dplyr by using dplyr} Iñaki Úcar shows surprise that some R users
|
|
1408
|
+
are trying to code in dplyr avoiding the use of NSE. For instance he
|
|
1409
|
+
says:
|
|
1410
|
+
|
|
1411
|
+
\begin{quote}
|
|
1412
|
+
Take the example of seplyr. It stands for standard evaluation dplyr, and
|
|
1413
|
+
enables us to program over dplyr without having ``to bring in (or study)
|
|
1414
|
+
any deep-theory or heavy-weight tools such as rlang/tidyeval''.
|
|
1415
|
+
\end{quote}
|
|
1416
|
+
|
|
1417
|
+
For me, there isn't really any surprise that users are trying to avoid
|
|
1418
|
+
dplyr deep-theory. R users frequently are not programmers and learning
|
|
1419
|
+
to code is already hard business, on top of that, having to learn how to
|
|
1420
|
+
`quote' or `enquo' or `quos' or `enquos' is not necessarily a `piece of
|
|
1421
|
+
cake'. So much so, that `tidyeval' has some more advanced functions that
|
|
1422
|
+
instead of using quoted expressions, uses strings as arguments.
|
|
1423
|
+
|
|
1424
|
+
In the following examples, we show the use of functions `group\_by\_at',
|
|
1425
|
+
`summarise\_at' and `rename\_at' that receive strings as argument. The
|
|
1426
|
+
data frame used in `starwars' that describes features of characters in
|
|
1427
|
+
the Starwars movies:
|
|
1428
|
+
|
|
1429
|
+
\begin{Shaded}
|
|
1430
|
+
\begin{Highlighting}[]
|
|
1431
|
+
\FunctionTok{puts}\NormalTok{ (}\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:starwars}\KeywordTok{]}\NormalTok{)}\AttributeTok{.head}
|
|
1432
|
+
\end{Highlighting}
|
|
1433
|
+
\end{Shaded}
|
|
1434
|
+
|
|
1435
|
+
\begin{verbatim}
|
|
1436
|
+
## # A tibble: 6 x 14
|
|
1437
|
+
## name height mass hair_color skin_color eye_color birth_year sex
|
|
1438
|
+
## <chr> <int> <dbl> <chr> <chr> <chr> <dbl> <chr>
|
|
1439
|
+
## 1 Luke ~ 172 77 blond fair blue 19 male
|
|
1440
|
+
## 2 C-3PO 167 75 <NA> gold yellow 112 none
|
|
1441
|
+
## 3 R2-D2 96 32 <NA> white, bl~ red 33 none
|
|
1442
|
+
## 4 Darth~ 202 136 none white yellow 41.9 male
|
|
1443
|
+
## 5 Leia ~ 150 49 brown light brown 19 fema~
|
|
1444
|
+
## 6 Owen ~ 178 120 brown, gr~ light blue 52 male
|
|
1445
|
+
## # i 6 more variables: gender <chr>, homeworld <chr>, species <chr>,
|
|
1446
|
+
## # films <list>, vehicles <list>, starships <list>
|
|
1447
|
+
\end{verbatim}
|
|
1448
|
+
|
|
1449
|
+
The grouped\_mean function below will receive a grouping variable and
|
|
1450
|
+
calculate summaries for the value\_variables given:
|
|
1451
|
+
|
|
1452
|
+
\begin{Shaded}
|
|
1453
|
+
\begin{Highlighting}[]
|
|
1454
|
+
\NormalTok{grouped\_mean }\OtherTok{\textless{}{-}} \ControlFlowTok{function}\NormalTok{(data, grouping\_variables, value\_variables) \{}
|
|
1455
|
+
\NormalTok{ data }\SpecialCharTok{\%\textgreater{}\%}
|
|
1456
|
+
\FunctionTok{group\_by\_at}\NormalTok{(grouping\_variables) }\SpecialCharTok{\%\textgreater{}\%}
|
|
1457
|
+
\FunctionTok{mutate}\NormalTok{(}\AttributeTok{count =} \FunctionTok{n}\NormalTok{()) }\SpecialCharTok{\%\textgreater{}\%}
|
|
1458
|
+
\FunctionTok{summarise\_at}\NormalTok{(}\FunctionTok{c}\NormalTok{(value\_variables, }\StringTok{"count"}\NormalTok{), mean, }\AttributeTok{na.rm =} \ConstantTok{TRUE}\NormalTok{) }\SpecialCharTok{\%\textgreater{}\%}
|
|
1459
|
+
\FunctionTok{rename\_at}\NormalTok{(value\_variables, }\FunctionTok{funs}\NormalTok{(}\FunctionTok{paste0}\NormalTok{(}\StringTok{"mean\_"}\NormalTok{, .)))}
|
|
1460
|
+
\NormalTok{ \}}
|
|
1461
|
+
|
|
1462
|
+
\NormalTok{gm }\OtherTok{=}\NormalTok{ starwars }\SpecialCharTok{\%\textgreater{}\%}
|
|
1463
|
+
\FunctionTok{grouped\_mean}\NormalTok{(}\StringTok{"eye\_color"}\NormalTok{, }\FunctionTok{c}\NormalTok{(}\StringTok{"mass"}\NormalTok{, }\StringTok{"birth\_year"}\NormalTok{))}
|
|
1464
|
+
\end{Highlighting}
|
|
1465
|
+
\end{Shaded}
|
|
1466
|
+
|
|
1467
|
+
\begin{verbatim}
|
|
1468
|
+
## Warning: `funs()` was deprecated in dplyr 0.8.0.
|
|
1469
|
+
## i Please use a list of either functions or lambdas:
|
|
1470
|
+
##
|
|
1471
|
+
## # Simple named list: list(mean = mean, median = median)
|
|
1472
|
+
##
|
|
1473
|
+
## # Auto named with `tibble::lst()`: tibble::lst(mean, median)
|
|
1474
|
+
##
|
|
1475
|
+
## # Using lambdas list(~ mean(., trim = .2), ~ median(., na.rm = TRUE))
|
|
1476
|
+
## Call `lifecycle::last_lifecycle_warnings()` to see where this warning
|
|
1477
|
+
## was generated.
|
|
1478
|
+
\end{verbatim}
|
|
1479
|
+
|
|
1480
|
+
\begin{Shaded}
|
|
1481
|
+
\begin{Highlighting}[]
|
|
1482
|
+
\FunctionTok{as.data.frame}\NormalTok{(gm) }
|
|
1483
|
+
\end{Highlighting}
|
|
1484
|
+
\end{Shaded}
|
|
1485
|
+
|
|
1486
|
+
\begin{verbatim}
|
|
1487
|
+
## eye_color mean_mass mean_birth_year count
|
|
1488
|
+
## 1 black 76.28571 33.00000 10
|
|
1489
|
+
## 2 blue 86.51667 67.06923 19
|
|
1490
|
+
## 3 blue-gray 77.00000 57.00000 1
|
|
1491
|
+
## 4 brown 66.09231 108.96429 21
|
|
1492
|
+
## 5 dark NaN NaN 1
|
|
1493
|
+
## 6 gold NaN NaN 1
|
|
1494
|
+
## 7 green, yellow 159.00000 NaN 1
|
|
1495
|
+
## 8 hazel 66.00000 34.50000 3
|
|
1496
|
+
## 9 orange 282.33333 231.00000 8
|
|
1497
|
+
## 10 pink NaN NaN 1
|
|
1498
|
+
## 11 red 81.40000 33.66667 5
|
|
1499
|
+
## 12 red, blue NaN NaN 1
|
|
1500
|
+
## 13 unknown 31.50000 NaN 3
|
|
1501
|
+
## 14 white 48.00000 NaN 1
|
|
1502
|
+
## 15 yellow 81.11111 76.38000 11
|
|
1503
|
+
\end{verbatim}
|
|
1504
|
+
|
|
1505
|
+
The same code with Galaaz, becomes:
|
|
1506
|
+
|
|
1507
|
+
\begin{Shaded}
|
|
1508
|
+
\begin{Highlighting}[]
|
|
1509
|
+
\ControlFlowTok{def}\NormalTok{ grouped\_mean(data, grouping\_variables, value\_variables)}
|
|
1510
|
+
\NormalTok{ data}\AttributeTok{.}
|
|
1511
|
+
\AttributeTok{group\_by\_at}\NormalTok{(grouping\_variables)}\AttributeTok{.}
|
|
1512
|
+
\AttributeTok{mutate}\NormalTok{(}\WarningTok{count:} \ConstantTok{E}\AttributeTok{.n}\NormalTok{)}\AttributeTok{.}
|
|
1513
|
+
\AttributeTok{summarise\_at}\NormalTok{(}
|
|
1514
|
+
\ConstantTok{E}\AttributeTok{.c}\NormalTok{(value\_variables, }\StringTok{"count"}\NormalTok{),}
|
|
1515
|
+
\ConstantTok{R}\KeywordTok{[}\WarningTok{:mean}\KeywordTok{]}\NormalTok{,}
|
|
1516
|
+
\WarningTok{na\_\_rm:} \DecValTok{true}\NormalTok{)}\AttributeTok{.}
|
|
1517
|
+
\AttributeTok{rename\_at}\NormalTok{(}
|
|
1518
|
+
\NormalTok{ value\_variables,}
|
|
1519
|
+
\ConstantTok{E}\AttributeTok{.funs}\NormalTok{(}\ConstantTok{E}\AttributeTok{.paste0}\NormalTok{(}\StringTok{"mean\_"}\NormalTok{, value\_variables)))}
|
|
1520
|
+
\ControlFlowTok{end}
|
|
1521
|
+
|
|
1522
|
+
\FunctionTok{puts}\NormalTok{ grouped\_mean(}
|
|
1523
|
+
\NormalTok{ (}\OperatorTok{\textasciitilde{}}\ConstantTok{R}\KeywordTok{[}\WarningTok{:starwars}\KeywordTok{]}\NormalTok{),}
|
|
1524
|
+
\StringTok{"eye\_color"}\NormalTok{,}
|
|
1525
|
+
\ConstantTok{E}\AttributeTok{.c}\NormalTok{(}\StringTok{"mass"}\NormalTok{, }\StringTok{"birth\_year"}\NormalTok{))}
|
|
1526
|
+
\end{Highlighting}
|
|
1527
|
+
\end{Shaded}
|
|
1528
|
+
|
|
1529
|
+
\begin{verbatim}
|
|
1530
|
+
## # A tibble: 15 x 4
|
|
1531
|
+
## eye_color mean_mass mean_birth_year count
|
|
1532
|
+
## <chr> <dbl> <dbl> <dbl>
|
|
1533
|
+
## 1 black 76.3 33 10
|
|
1534
|
+
## 2 blue 86.5 67.1 19
|
|
1535
|
+
## 3 blue-gray 77 57 1
|
|
1536
|
+
## 4 brown 66.1 109. 21
|
|
1537
|
+
## 5 dark NaN NaN 1
|
|
1538
|
+
## 6 gold NaN NaN 1
|
|
1539
|
+
## 7 green, yellow 159 NaN 1
|
|
1540
|
+
## 8 hazel 66 34.5 3
|
|
1541
|
+
## 9 orange 282. 231 8
|
|
1542
|
+
## 10 pink NaN NaN 1
|
|
1543
|
+
## 11 red 81.4 33.7 5
|
|
1544
|
+
## 12 red, blue NaN NaN 1
|
|
1545
|
+
## 13 unknown 31.5 NaN 3
|
|
1546
|
+
## 14 white 48 NaN 1
|
|
1547
|
+
## 15 yellow 81.1 76.4 11
|
|
1548
|
+
\end{verbatim}
|
|
1549
|
+
|
|
1550
|
+
\section{Further reading}\label{further-reading}
|
|
1551
|
+
|
|
1552
|
+
\begin{itemize}
|
|
1553
|
+
\tightlist
|
|
1554
|
+
\item
|
|
1555
|
+
\href{https://www.jruby.org/}{JRuby} --- Ruby on the JVM (Galaaz 2.0)
|
|
1556
|
+
\item
|
|
1557
|
+
\href{https://medium.freecodecamp.org/how-to-make-beautiful-ruby-plots-with-galaaz-320848058857}{How
|
|
1558
|
+
to make Beautiful Ruby Plots with Galaaz} (plots; narrative partly
|
|
1559
|
+
pre-2.0)
|
|
1560
|
+
\item
|
|
1561
|
+
\href{https://towardsdatascience.com/ruby-plotting-with-galaaz-an-example-of-tightly-coupling-ruby-and-r-in-graalvm-520b69e21021}{Ruby
|
|
1562
|
+
Plotting with Galaaz in GraalVM} (older stack; ideas still useful)
|
|
1563
|
+
\item
|
|
1564
|
+
\href{https://towardsdatascience.com/how-to-do-reproducible-research-in-ruby-with-gknit-c26d2684d64e}{How
|
|
1565
|
+
to do reproducible research in Ruby with gKnit}
|
|
1566
|
+
\item
|
|
1567
|
+
\href{https://r4ds.had.co.nz/}{R for Data Science}
|
|
1568
|
+
\item
|
|
1569
|
+
\href{https://adv-r.hadley.nz/}{Advanced R}
|
|
1570
|
+
\item
|
|
1571
|
+
Historical context: \href{https://www.graalvm.org/}{GraalVM},
|
|
1572
|
+
\href{https://github.com/oracle/truffleruby}{TruffleRuby},
|
|
1573
|
+
\href{https://github.com/oracle/fastr}{FastR}
|
|
1574
|
+
\end{itemize}
|
|
1575
|
+
|
|
1576
|
+
\section{Conclusion}\label{conclusion}
|
|
1577
|
+
|
|
1578
|
+
Ruby and Galaaz provide a nice framework for developing code that uses R
|
|
1579
|
+
functions. Although R is a very powerful and flexible language,
|
|
1580
|
+
sometimes, too much flexibility makes life harder for the casual user.
|
|
1581
|
+
We believe however, that even for the advanced user, Ruby integrated
|
|
1582
|
+
with R throught Galaaz, makes a powerful environment for data analysis.
|
|
1583
|
+
In this blog post we showed how Galaaz consistent syntax eliminates the
|
|
1584
|
+
need for complex constructs such as quoting, enquoting, quasiquotation,
|
|
1585
|
+
etc. This simplification comes from the fact that expressions and
|
|
1586
|
+
variables are clearly separated objects, which is not the case in the R
|
|
1587
|
+
language.
|
|
1588
|
+
|
|
1589
|
+
\end{document}
|