galaaz 0.4.10 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (391) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +26 -0
  3. data/LICENSE +0 -0
  4. data/README.md +3123 -882
  5. data/Rakefile +62 -41
  6. data/bin/galaaz-bootstrap +137 -0
  7. data/bin/galaaz-jruby +14 -0
  8. data/bin/galaaz_jruby_env.inc.sh +6 -0
  9. data/bin/gbookdown +64 -0
  10. data/bin/gknit +223 -6
  11. data/bin/gknit-draft +105 -0
  12. data/bin/gknit-draft.rb +28 -0
  13. data/bin/gknit_Rscript +127 -0
  14. data/bin/grun +27 -1
  15. data/bin/gstudio +49 -4
  16. data/bin/{gstudio.rb → gstudio_irb.rb} +0 -0
  17. data/bin/gstudio_pry.rb +7 -0
  18. data/bin/install-tinytex +6 -0
  19. data/bin/run_all_rspec +43 -0
  20. data/bin/run_example +14 -0
  21. data/bin/run_old_rspec +19 -0
  22. data/bin/run_rspec +23 -0
  23. data/bin/run_rspec_subset +38 -0
  24. data/bin/run_slow_rspec +19 -0
  25. data/blogs/R-on-Rails-Planning-Document.md +940 -0
  26. data/blogs/README.md +100 -0
  27. data/blogs/galaaz_ggplot/galaaz_ggplot.Rmd +38 -66
  28. data/blogs/galaaz_ggplot/galaaz_ggplot.log +754 -0
  29. data/blogs/galaaz_ggplot/galaaz_ggplot.md +364 -0
  30. data/blogs/galaaz_ggplot/galaaz_ggplot.tex +607 -0
  31. data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-html/midwest_rb.png +0 -0
  32. data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-html/scatter_plot_rb.png +0 -0
  33. data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-markdown_github/midwest_rb.png +0 -0
  34. data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-markdown_github/scatter_plot_rb.png +0 -0
  35. data/blogs/galaaz_ggplot/midwest.Rmd +3 -3
  36. data/blogs/galaaz_ggplot/midwest_external_png +0 -0
  37. data/blogs/gknit/gknit.Rmd +52 -55
  38. data/blogs/gknit/gknit.md +94 -94
  39. data/blogs/gknit/gknit_files/figure-html/bubble-1.png +0 -0
  40. data/blogs/gknit/gknit_files/figure-html/diverging_bar.png +0 -0
  41. data/blogs/gknit/lst.rds +0 -0
  42. data/blogs/gknit/model.rb +1 -1
  43. data/blogs/gknit/stats.bib +0 -0
  44. data/blogs/manual/include_model_local_repro.Rmd +14 -0
  45. data/blogs/manual/include_model_local_repro.md +75 -0
  46. data/blogs/manual/lst.rds +0 -0
  47. data/blogs/manual/manual.Rmd +1582 -196
  48. data/blogs/manual/manual.log +1786 -0
  49. data/blogs/manual/manual.md +3107 -890
  50. data/blogs/manual/manual.tex +3018 -1086
  51. data/blogs/manual/manual_files/figure-html/bubble-1.png +0 -0
  52. data/blogs/manual/manual_files/figure-html/diverging_bar.png +0 -0
  53. data/blogs/manual/manual_files/figure-latex/bubble-1.png +0 -0
  54. data/blogs/manual/model.rb +41 -0
  55. data/blogs/nse_dplyr/nse_dplyr.Rmd +277 -151
  56. data/blogs/nse_dplyr/nse_dplyr.log +928 -0
  57. data/blogs/nse_dplyr/nse_dplyr.md +457 -293
  58. data/blogs/oh_my/not_so.rb +0 -0
  59. data/blogs/oh_my/oh_my.Rmd +1234 -25
  60. data/blogs/oh_my/oh_my.log +804 -0
  61. data/blogs/oh_my/oh_my.md +1808 -228
  62. data/blogs/oh_my/oh_my.tex +821 -0
  63. data/blogs/oh_my/old.Rmd +15 -14
  64. data/blogs/ruby_plot/ruby_plot.Rmd +58 -82
  65. data/blogs/ruby_plot/ruby_plot.log +885 -0
  66. data/blogs/ruby_plot/ruby_plot.md +71 -103
  67. data/blogs/ruby_plot/ruby_plot.tex +940 -0
  68. data/blogs/ruby_plot/ruby_plot_files/figure-html/dose_len.png +0 -0
  69. data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_delivery.png +0 -0
  70. data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_dose.png +0 -0
  71. data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color.png +0 -0
  72. data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color2.png +0 -0
  73. data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_decorations.png +0 -0
  74. data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_jitter.png +0 -0
  75. data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_points.png +0 -0
  76. data/blogs/ruby_plot/ruby_plot_files/figure-html/final_box_plot.png +0 -0
  77. data/blogs/ruby_plot/ruby_plot_files/figure-html/final_violin_plot.png +0 -0
  78. data/blogs/ruby_plot/ruby_plot_files/figure-html/violin_with_jitter.png +0 -0
  79. data/blogs/ruby_plot/ruby_plot_files/figure-latex/dose_len.png +0 -0
  80. data/blogs/ruby_plot/ruby_plot_files/figure-latex/facet_by_delivery.png +0 -0
  81. data/blogs/ruby_plot/ruby_plot_files/figure-latex/facet_by_dose.png +0 -0
  82. data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_by_delivery_color.png +0 -0
  83. data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_by_delivery_color2.png +0 -0
  84. data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_decorations.png +0 -0
  85. data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_jitter.png +0 -0
  86. data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_points.png +0 -0
  87. data/blogs/ruby_plot/ruby_plot_files/figure-latex/final_box_plot.png +0 -0
  88. data/blogs/ruby_plot/ruby_plot_files/figure-latex/final_violin_plot.png +0 -0
  89. data/blogs/ruby_plot/ruby_plot_files/figure-latex/violin_with_jitter.png +0 -0
  90. data/blogs/test/test.Rmd +14 -0
  91. data/examples/50Plots_MasterList/Images/midwest-scatterplot.PNG +0 -0
  92. data/examples/50Plots_MasterList/ScatterPlot.rb +0 -0
  93. data/examples/50Plots_MasterList/scatter_plot.rb +0 -0
  94. data/examples/Bibliography/master.bib +50 -0
  95. data/examples/Bibliography/stats.bib +72 -0
  96. data/examples/R/calc.R +0 -0
  97. data/examples/R/java_interop.R +0 -0
  98. data/examples/bioconductor_deseq2_airway/Documentation/DESeq2-airway-walkthrough.md +56 -0
  99. data/examples/bioconductor_deseq2_airway/bench_galaaz_three_same_process.rb +53 -0
  100. data/examples/bioconductor_deseq2_airway/bench_r_three_same_process.R +34 -0
  101. data/examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb +33 -0
  102. data/examples/bioconductor_deseq2_airway/deseq2_airway_galaaz_optimized.rb +34 -0
  103. data/examples/bioconductor_deseq2_airway/deseq2_airway_minimal.R +30 -0
  104. data/examples/bioconductor_deseq2_airway/deseq2_airway_pipeline_for_bench.R +36 -0
  105. data/examples/islr/all.rb +13 -0
  106. data/examples/islr/ch2.spec.rb +37 -7
  107. data/examples/islr/ch3.spec.rb +11 -2
  108. data/examples/islr/ch3_boston.rb +27 -0
  109. data/examples/islr/ch3_multiple_regression.rb +0 -0
  110. data/examples/islr/ch6.spec.rb +24 -1
  111. data/examples/islr/x_y_rnorm.jpg +0 -0
  112. data/examples/latex_templates/Test-acm_article/Makefile +16 -0
  113. data/examples/latex_templates/Test-acm_article/Test-acm_article.Rmd +65 -0
  114. data/examples/latex_templates/Test-acm_article/acm_proc_article-sp.cls +1670 -0
  115. data/examples/latex_templates/Test-acm_article/sensys-abstract.cls +703 -0
  116. data/examples/latex_templates/Test-acm_article/sigproc.bib +59 -0
  117. data/examples/latex_templates/Test-acs_article/Test-acs_article.Rmd +260 -0
  118. data/examples/latex_templates/Test-acs_article/acs-Test-acs_article.bib +11 -0
  119. data/examples/latex_templates/Test-acs_article/acs-my_output.bib +11 -0
  120. data/examples/latex_templates/Test-acs_article/acstest.bib +17 -0
  121. data/examples/latex_templates/Test-aea_article/AEA.cls +1414 -0
  122. data/{blogs/gknit/marshal.dump → examples/latex_templates/Test-aea_article/BibFile.bib} +0 -0
  123. data/examples/latex_templates/Test-aea_article/Test-aea_article.Rmd +108 -0
  124. data/examples/latex_templates/Test-aea_article/aea.bst +1269 -0
  125. data/examples/latex_templates/Test-aea_article/multicol.sty +853 -0
  126. data/examples/latex_templates/Test-aea_article/references.bib +0 -0
  127. data/examples/latex_templates/Test-aea_article/setspace.sty +546 -0
  128. data/examples/latex_templates/Test-amq_article/Test-amq_article.Rmd +256 -0
  129. data/examples/latex_templates/Test-amq_article/Test-amq_article.pdfsync +3397 -0
  130. data/examples/latex_templates/Test-ams_article/Test-ams_article.Rmd +215 -0
  131. data/examples/latex_templates/Test-ams_article/amstest.bib +436 -0
  132. data/examples/latex_templates/Test-asa_article/Test-asa_article.Rmd +153 -0
  133. data/examples/latex_templates/Test-asa_article/agsm.bst +1353 -0
  134. data/examples/latex_templates/Test-asa_article/bibliography.bib +233 -0
  135. data/examples/latex_templates/Test-ieee_article/IEEEtran.bst +2409 -0
  136. data/examples/latex_templates/Test-ieee_article/IEEEtran.cls +6346 -0
  137. data/examples/latex_templates/Test-ieee_article/Test-ieee_article.Rmd +175 -0
  138. data/examples/latex_templates/Test-ieee_article/mybibfile.bib +20 -0
  139. data/examples/latex_templates/Test-rjournal_article/RJournal.sty +335 -0
  140. data/examples/latex_templates/Test-rjournal_article/RJreferences.bib +18 -0
  141. data/examples/latex_templates/Test-rjournal_article/Test-rjournal_article.Rmd +52 -0
  142. data/examples/latex_templates/Test-springer_article/Test-springer_article.Rmd +65 -0
  143. data/examples/latex_templates/Test-springer_article/bibliography.bib +26 -0
  144. data/examples/latex_templates/Test-springer_article/spbasic.bst +1658 -0
  145. data/examples/latex_templates/Test-springer_article/spmpsci.bst +1512 -0
  146. data/examples/latex_templates/Test-springer_article/spphys.bst +1443 -0
  147. data/examples/latex_templates/Test-springer_article/svglov3.clo +113 -0
  148. data/examples/latex_templates/Test-springer_article/svjour3.cls +1431 -0
  149. data/examples/misc/baseball.csv +0 -0
  150. data/examples/misc/ggplot.rb +3 -2
  151. data/examples/misc/moneyball.rb +0 -0
  152. data/examples/misc/subsetting.rb +0 -0
  153. data/examples/multithread_shards_to_r/shards_to_r.rb +67 -0
  154. data/examples/rmarkdown/svm-rmarkdown-anon-ms-example/svm-rmarkdown-anon-ms-example.Rmd +73 -0
  155. data/examples/rmarkdown/svm-rmarkdown-article-example/svm-rmarkdown-article-example.Rmd +382 -0
  156. data/examples/rmarkdown/svm-rmarkdown-beamer-example/svm-rmarkdown-beamer-example.Rmd +164 -0
  157. data/examples/rmarkdown/svm-rmarkdown-cv/svm-rmarkdown-cv.Rmd +92 -0
  158. data/examples/rmarkdown/svm-rmarkdown-syllabus-example/attend-grade-relationships.csv +482 -0
  159. data/examples/rmarkdown/svm-rmarkdown-syllabus-example/svm-rmarkdown-syllabus-example.Rmd +280 -0
  160. data/examples/rmarkdown/svm-xaringan-example/svm-xaringan-example.Rmd +386 -0
  161. data/examples/sthda_ggplot/README.md +0 -0
  162. data/examples/sthda_ggplot/RUN.md +41 -0
  163. data/examples/sthda_ggplot/all.rb +0 -0
  164. data/examples/sthda_ggplot/one_variable_continuous/density_gg.rb +0 -0
  165. data/examples/sthda_ggplot/one_variable_continuous/geom_area.rb +0 -0
  166. data/examples/sthda_ggplot/one_variable_continuous/geom_density.rb +2 -0
  167. data/examples/sthda_ggplot/one_variable_continuous/geom_dotplot.rb +0 -0
  168. data/examples/sthda_ggplot/one_variable_continuous/geom_freqpoly.rb +0 -0
  169. data/examples/sthda_ggplot/one_variable_continuous/geom_histogram.rb +0 -0
  170. data/examples/sthda_ggplot/one_variable_continuous/histogram_density.rb +0 -0
  171. data/examples/sthda_ggplot/one_variable_continuous/stat.rb +0 -0
  172. data/examples/sthda_ggplot/one_variable_discrete/bar.rb +0 -0
  173. data/examples/sthda_ggplot/qplots/box_violin_dot.rb +0 -0
  174. data/examples/sthda_ggplot/qplots/scatter_plots.rb +0 -0
  175. data/examples/sthda_ggplot/scatter_gg.rb +0 -0
  176. data/examples/sthda_ggplot/two_variables_cont_bivariate/geom_bin2d.rb +0 -0
  177. data/examples/sthda_ggplot/two_variables_cont_bivariate/geom_density2d.rb +0 -0
  178. data/examples/sthda_ggplot/two_variables_cont_bivariate/geom_hex.rb +0 -0
  179. data/examples/sthda_ggplot/two_variables_cont_cont/geom_point.rb +0 -0
  180. data/examples/sthda_ggplot/two_variables_cont_cont/geom_smooth.rb +0 -0
  181. data/examples/sthda_ggplot/two_variables_cont_cont/misc.rb +0 -0
  182. data/examples/sthda_ggplot/two_variables_cont_function/geom_area.rb +4 -3
  183. data/examples/sthda_ggplot/two_variables_disc_cont/geom_bar.rb +0 -0
  184. data/examples/sthda_ggplot/two_variables_disc_cont/geom_boxplot.rb +0 -0
  185. data/examples/sthda_ggplot/two_variables_disc_cont/geom_dotplot.rb +0 -0
  186. data/examples/sthda_ggplot/two_variables_disc_cont/geom_jitter.rb +0 -0
  187. data/examples/sthda_ggplot/two_variables_disc_cont/geom_line.rb +0 -0
  188. data/examples/sthda_ggplot/two_variables_disc_cont/geom_violin.rb +0 -0
  189. data/examples/sthda_ggplot/two_variables_disc_disc/geom_jitter.rb +0 -0
  190. data/examples/sthda_ggplot/two_variables_error/geom_crossbar.rb +0 -0
  191. data/ext/new_bridge/Makefile +46 -0
  192. data/ext/new_bridge/galaaz_gatekeeper_phase0.cpp +12 -0
  193. data/ext/new_bridge/galaaz_gatekeeper_phase1.cpp +1639 -0
  194. data/lib/R_interface/galaaz_device.R +20 -0
  195. data/lib/R_interface/include_engine.R +109 -0
  196. data/lib/R_interface/new_bridge_adapter.rb +824 -0
  197. data/lib/R_interface/r.rb +177 -25
  198. data/lib/R_interface/r_arrow.rb +113 -0
  199. data/lib/R_interface/r_libs.R +4 -4
  200. data/lib/R_interface/r_methods.rb +13 -116
  201. data/lib/R_interface/r_module_s.rb +0 -0
  202. data/lib/R_interface/rbinary_operators.rb +20 -2
  203. data/lib/R_interface/rclosure.rb +5 -1
  204. data/lib/R_interface/rdata_frame.rb +34 -70
  205. data/lib/R_interface/rdevice.rb +125 -0
  206. data/lib/R_interface/rdevices.R +0 -0
  207. data/lib/R_interface/renvironment.rb +10 -4
  208. data/lib/R_interface/rexpression.rb +5 -1
  209. data/lib/R_interface/rindexed_object.rb +41 -13
  210. data/lib/R_interface/rlanguage.rb +20 -62
  211. data/lib/R_interface/rlist.rb +115 -25
  212. data/lib/R_interface/rlogical_operators.rb +0 -0
  213. data/lib/R_interface/rmatrix.rb +2 -11
  214. data/lib/R_interface/rmd_indexed_object.rb +5 -1
  215. data/lib/R_interface/robject.rb +348 -290
  216. data/lib/R_interface/rpkg.rb +1 -0
  217. data/lib/R_interface/rsupport.rb +610 -331
  218. data/lib/R_interface/rsupport_scope.rb +2 -1
  219. data/lib/R_interface/rsymbol.rb +50 -0
  220. data/lib/R_interface/ruby_callback.rb +2 -3
  221. data/lib/R_interface/ruby_extensions.rb +225 -175
  222. data/lib/R_interface/runary_operators.rb +0 -0
  223. data/lib/R_interface/rvector.rb +147 -31
  224. data/lib/galaaz.rb +0 -0
  225. data/lib/galaaz_jruby.rb +22 -0
  226. data/lib/gknit/diagnostics.rb +50 -0
  227. data/lib/gknit/draft.rb +111 -0
  228. data/lib/gknit/include_engine.rb +15 -7
  229. data/lib/gknit/knitr_engine.rb +223 -107
  230. data/lib/gknit/rb_engine.rb +3 -3
  231. data/lib/gknit/ruby_engine.rb +0 -0
  232. data/lib/gknit.rb +3 -0
  233. data/lib/new_bridge/bootstrap/windows_bootstrap.rb +285 -0
  234. data/lib/new_bridge/envelope.rb +51 -0
  235. data/lib/new_bridge/eval_result.rb +26 -0
  236. data/lib/new_bridge/framing.rb +39 -0
  237. data/lib/new_bridge/instance_pool_client.rb +38 -0
  238. data/lib/new_bridge/r_instance_manager.rb +404 -0
  239. data/lib/new_bridge/session_client.rb +530 -0
  240. data/lib/new_bridge/tcp_framed.rb +44 -0
  241. data/lib/new_bridge.rb +9 -0
  242. data/lib/util/exec_ruby.rb +95 -46
  243. data/lib/util/inline_file.rb +35 -30
  244. data/new_bridge_specs/benchmark_phase5_5_unboxing_spec.rb +96 -0
  245. data/new_bridge_specs/eval_r_async_spec.rb +113 -0
  246. data/new_bridge_specs/integration_phase5_1_concurrent_spec.rb +50 -0
  247. data/new_bridge_specs/integration_phase5_1_eval_spec.rb +16 -0
  248. data/new_bridge_specs/integration_phase5_1_r_api_spec.rb +25 -0
  249. data/new_bridge_specs/integration_phase5_1_smoke_spec.rb +31 -0
  250. data/new_bridge_specs/integration_phase5_2_dataframe_unboxing_spec.rb +19 -0
  251. data/new_bridge_specs/integration_phase5_2_handle_eval_unboxing_spec.rb +25 -0
  252. data/new_bridge_specs/integration_phase5_3_callback_args_spec.rb +28 -0
  253. data/new_bridge_specs/integration_phase5_3_callback_error_spec.rb +22 -0
  254. data/new_bridge_specs/integration_phase5_3_callback_timeout_spec.rb +28 -0
  255. data/new_bridge_specs/integration_phase5_3_callbacks_smoke_spec.rb +22 -0
  256. data/new_bridge_specs/integration_phase5_3_edge_cases_spec.rb +52 -0
  257. data/new_bridge_specs/integration_phase5_3_nested_spec.rb +30 -0
  258. data/new_bridge_specs/integration_phase5_4_concurrent_sessions_spec.rb +53 -0
  259. data/new_bridge_specs/integration_phase5_4_nested_session_callbacks_spec.rb +49 -0
  260. data/new_bridge_specs/integration_phase5_4_session_routing_spec.rb +38 -0
  261. data/new_bridge_specs/integration_phase5_5_stress_concurrency_spec.rb +52 -0
  262. data/new_bridge_specs/integration_phase5_5_unbox_walk_spec.rb +46 -0
  263. data/new_bridge_specs/phase0_protocol_spec.rb +96 -0
  264. data/new_bridge_specs/phase1_req_ret_spec.rb +66 -0
  265. data/new_bridge_specs/phase2_multi_instance_spec.rb +67 -0
  266. data/new_bridge_specs/phase3_callbacks_spec.rb +71 -0
  267. data/new_bridge_specs/phase4_2_hardening_spec.rb +252 -0
  268. data/new_bridge_specs/phase4_3_r_instance_manager_spec.rb +85 -0
  269. data/new_bridge_specs/phase4_nested_callbacks_spec.rb +123 -0
  270. data/r_requires/ggplot.rb +0 -0
  271. data/r_requires/knitr.rb +0 -0
  272. data/specs/all.rb +15 -11
  273. data/specs/arrow_from_ruby_batches_spec.rb +50 -0
  274. data/specs/arrow_semantics_spec.rb +64 -0
  275. data/specs/bridge_concurrent_spec.rb +46 -0
  276. data/specs/bridge_nested_spec.rb +25 -0
  277. data/specs/dataframe_semantics_spec.rb +122 -0
  278. data/specs/dataframe_single_index_logical_filter_spec.rb +21 -0
  279. data/specs/dispatch_probe_cache_spec.rb +38 -0
  280. data/specs/dispatch_probe_error_class_fallback_spec.rb +20 -0
  281. data/specs/dispatch_probe_fallback_spec.rb +18 -0
  282. data/specs/environment_semantics_spec.rb +89 -0
  283. data/specs/field_access_spec.rb +31 -0
  284. data/specs/figures/bg.jpeg +0 -0
  285. data/specs/figures/bg.png +0 -0
  286. data/specs/figures/bg.svg +168 -57
  287. data/specs/figures/dose_len.png +0 -0
  288. data/specs/figures/no_args.jpeg +0 -0
  289. data/specs/figures/no_args.png +0 -0
  290. data/specs/figures/no_args.svg +168 -57
  291. data/specs/figures/width_height.jpeg +0 -0
  292. data/specs/figures/width_height.png +0 -0
  293. data/specs/figures/width_height_units1.jpeg +0 -0
  294. data/specs/figures/width_height_units1.png +0 -0
  295. data/specs/figures/width_height_units2.jpeg +0 -0
  296. data/specs/figures/width_height_units2.png +0 -0
  297. data/specs/formula_semantics_spec.rb +81 -0
  298. data/specs/galaaz_util_exec_ruby_spec.rb +85 -0
  299. data/specs/galaaz_util_inline_file_spec.rb +54 -0
  300. data/specs/gknit_cli_option_permutation_spec.rb +24 -0
  301. data/specs/gknit_include_engine_spec.rb +72 -0
  302. data/specs/gknit_install_timeout_report_spec.rb +69 -0
  303. data/specs/gknit_internal_error_report_spec.rb +57 -0
  304. data/specs/gknit_vector_map_output_spec.rb +59 -0
  305. data/specs/globalenv_guardrail_spec.rb +52 -0
  306. data/specs/language_expression_semantics_spec.rb +145 -0
  307. data/specs/list_semantics_spec.rb +111 -0
  308. data/specs/new_bridge_bulk_dataframe_transfer_spec.rb +44 -0
  309. data/specs/new_bridge_bulk_vector_transfer_spec.rb +73 -0
  310. data/specs/new_bridge_callback_timeout_spec.rb +69 -0
  311. data/specs/new_bridge_eval_r_fallback_spec.rb +55 -0
  312. data/specs/nil_null_spec.rb +42 -0
  313. data/specs/object_build_phase2_spec.rb +53 -0
  314. data/specs/phase1_callback_bridge_spec.rb +84 -0
  315. data/specs/phase2_gknit_generic_rendering_guardrail_spec.rb +46 -0
  316. data/specs/phase2_gknit_no_raw_code_leakage_spec.rb +43 -0
  317. data/specs/phase3_gknit_generic_graphics_capture_spec.rb +71 -0
  318. data/specs/plot_device_semantics_spec.rb +28 -0
  319. data/specs/plot_snapshot_semantics_spec.rb +58 -0
  320. data/specs/protocol_result_spec.rb +236 -0
  321. data/specs/r_batch_fail_fast_spec.rb +47 -0
  322. data/specs/r_bridge_bootstrap_spec.rb +11 -0
  323. data/specs/r_devices.spec.rb +1 -1
  324. data/specs/r_eval.spec.rb +16 -18
  325. data/specs/r_function.spec.rb +1 -1
  326. data/specs/r_instance_manager_spec.rb +285 -0
  327. data/specs/r_list_apply.spec.rb +15 -15
  328. data/specs/r_matrix.spec.rb +0 -0
  329. data/specs/r_nse.spec.rb +5 -5
  330. data/specs/r_object_send_dispatch_spec.rb +13 -0
  331. data/specs/r_vector_comparator_spec.rb +8 -0
  332. data/specs/r_vector_creation.spec.rb +0 -0
  333. data/specs/r_vector_functions.spec.rb +0 -0
  334. data/specs/r_vector_object.spec.rb +0 -0
  335. data/specs/r_vector_operators.spec.rb +0 -0
  336. data/specs/r_vector_structured_scalar_reads_spec.rb +35 -0
  337. data/specs/r_vector_subsetting.spec.rb +0 -0
  338. data/specs/range_helper_spec.rb +21 -0
  339. data/specs/rsupport_scope_spec.rb +28 -0
  340. data/specs/rsupport_var_name_thread_safety_spec.rb +24 -0
  341. data/specs/scalar_character_spec.rb +44 -0
  342. data/specs/scoped_symbol_dsl_refinement_spec.rb +40 -0
  343. data/specs/session_env_bridge_spec.rb +25 -0
  344. data/specs/simplecov_bootstrap_spec.rb +10 -0
  345. data/specs/spec_helper.rb +10 -0
  346. data/specs/tmp.rb +41 -20
  347. data/specs/unboxing_recursion_regression_spec.rb +30 -0
  348. data/specs/unboxing_spec.rb +49 -0
  349. data/specs/verify_callbacks.rb +42 -0
  350. data/sty/galaaz.sty +0 -0
  351. data/version.rb +1 -1
  352. metadata +239 -71
  353. data/blogs/galaaz_ggplot/galaaz_ggplot.aux +0 -41
  354. data/blogs/galaaz_ggplot/galaaz_ggplot.html +0 -705
  355. data/blogs/galaaz_ggplot/galaaz_ggplot.out +0 -10
  356. data/blogs/galaaz_ggplot/galaaz_ggplot.pdf +0 -0
  357. data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-latex/midwest_rb.pdf +0 -0
  358. data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-latex/scatter_plot_rb.pdf +0 -0
  359. data/blogs/galaaz_ggplot/midwest.html +0 -188
  360. data/blogs/gknit/gknit.html +0 -2266
  361. data/blogs/gknit/gknit.pdf +0 -0
  362. data/blogs/gknit/gknit.tex +0 -1358
  363. data/blogs/manual/graph.rb +0 -29
  364. data/blogs/manual/manual.html +0 -2995
  365. data/blogs/manual/manual.pdf +0 -0
  366. data/blogs/manual/manual_files/figure-latex/diverging_bar.pdf +0 -0
  367. data/blogs/nse_dplyr/nse_dplyr.html +0 -960
  368. data/blogs/nse_dplyr/nse_dplyr.pdf +0 -0
  369. data/blogs/nse_dplyr/nse_dplyr.tex +0 -1373
  370. data/blogs/oh_my/oh_my.html +0 -680
  371. data/blogs/ruby_plot/ruby_plot.Rmd_external_figs +0 -662
  372. data/blogs/ruby_plot/ruby_plot.html +0 -729
  373. data/blogs/ruby_plot/ruby_plot.pdf +0 -0
  374. data/blogs/ruby_plot/ruby_plot_files/figure-html/dose_len.svg +0 -57
  375. data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_delivery.svg +0 -106
  376. data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_dose.svg +0 -110
  377. data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color.svg +0 -174
  378. data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color2.svg +0 -236
  379. data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_jitter.svg +0 -296
  380. data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_points.svg +0 -236
  381. data/blogs/ruby_plot/ruby_plot_files/figure-html/final_box_plot.svg +0 -218
  382. data/blogs/ruby_plot/ruby_plot_files/figure-html/final_violin_plot.svg +0 -128
  383. data/blogs/ruby_plot/ruby_plot_files/figure-html/violin_with_jitter.svg +0 -150
  384. data/examples/paper/paper.rb +0 -36
  385. data/specs/r_dataframe.spec.rb +0 -379
  386. data/specs/r_environment.spec.rb +0 -140
  387. data/specs/r_formula.spec.rb +0 -232
  388. data/specs/r_language.spec.rb +0 -112
  389. data/specs/r_list.spec.rb +0 -293
  390. data/specs/r_plots.spec.rb +0 -72
  391. data/specs/ruby_expression.spec.rb +0 -315
@@ -1,26 +1,34 @@
1
1
  ---
2
2
  title: "Galaaz Manual"
3
- subtitle: "How to tightly couple Ruby and R in GraalVM"
3
+ subtitle: "Coupling Ruby (JRuby) and GNU R for data science"
4
4
  author: "Rodrigo Botafogo"
5
- tags: [Galaaz, Ruby, R, TruffleRuby, FastR, GraalVM, ggplot2]
6
- date: "2019"
5
+ tags: [Galaaz, Ruby, JRuby, R, "GNU R", ggplot2, knitr, dplyr, Bioconductor, Arrow]
6
+ date: "2026"
7
+ bibliography: "../../examples/Bibliography/stats.bib"
7
8
  output:
9
+ html_document:
10
+ self_contained: true
11
+ keep_md: true
12
+ toc: true
13
+ toc_depth: 3
14
+ number_sections: true
8
15
  pdf_document:
9
16
  includes:
10
17
  in_header: "../../sty/galaaz.sty"
11
18
  keep_tex: yes
12
19
  number_sections: yes
13
20
  toc: true
14
- toc_depth: 2
15
- html_document:
16
- self_contained: true
17
- keep_md: true
21
+ toc_depth: 3
18
22
  md_document:
19
23
  variant: markdown_github
20
24
  fontsize: 11pt
21
25
  ---
22
26
 
23
27
  ```{ruby setup, echo=FALSE}
28
+ # Bridge default is 60s; some chunks (Arrow, large dplyr pipes) need more.
29
+ ENV['GALAAZ_BRIDGE_TIMEOUT_SEC'] ||= '300'
30
+
31
+ R.options(crayon__enabled: false)
24
32
  R.install_and_loads('kableExtra')
25
33
  ```
26
34
 
@@ -30,30 +38,272 @@ Galaaz is a system for tightly coupling Ruby and R. Ruby is a powerful language,
30
38
  community, a very large set of libraries and great for web development. However, it lacks
31
39
  libraries for data science, statistics, scientific plotting and machine learning. On the
32
40
  other hand, R is considered one of the most powerful languages for solving all of the above
33
- problems. Maybe the strongest competitor to R is Python with libraries such as NumPy,
34
- Panda, SciPy, SciKit-Learn and a couple more.
41
+ problems. **Python** is a strong competitor: NumPy, pandas, SciPy, and scikit-learn are
42
+ widely used building blocks, and **PyPI** hosts many thousands of other packages for
43
+ numerical work, machine learning, and beyond.
44
+
45
+ With Galaaz we do not intend to re-implement any of the scientific libraries in R, we allow
46
+ for very tight coupling between the two languages to the point that the Ruby developer does
47
+ not need to know that there is an R engine running.
48
+
49
+ According to Wikipedia "Ruby is a dynamic, interpreted, reflective, object-oriented,
50
+ general-purpose programming language. It was designed and developed in the mid-1990s by Yukihiro
51
+ "Matz" Matsumoto in Japan." It reached high popularity with the development of Ruby on Rails
52
+ (RoR) by David Heinemeier Hansson. RoR is a web application framework first released
53
+ around 2005. It makes extensive use of Ruby's metaprogramming features. With RoR,
54
+ Ruby became very popular. According to [Ruby’s place in the TIOBE index](https://www.tiobe.com/tiobe-index/ruby/)
55
+ it peaked in popularity around 2008, then declined until 2015 when it started picking up again.
56
+ Ruby remains a significant language in web development and general-purpose scripting.
57
+
58
+ Python, a language similar to Ruby, ranks 4th in the index. Java, C and C++ take the
59
+ first three positions. Ruby is often criticized for its focus on web applications.
60
+ But Ruby can do [much more](https://github.com/markets/awesome-ruby) than just web applications.
61
+ Yet, for scientific computing, Ruby lags behind Python and R. Python offers Django and
62
+ similar frameworks for the web, plus NumPy, pandas, and a deep catalog of science and ML libraries.
63
+ R is a free software environment for statistical computing and graphics with thousands
64
+ of libraries for data analysis.
65
+
66
+ Until recently, there was no real perspective for Ruby to bridge this gap.
67
+ Implementing a complete scientific computing infrastructure would take too long.
68
+
69
+ **Galaaz 2.0** couples **JRuby** (Ruby on the JVM) with **GNU R**—the same R you use for
70
+ CRAN and Bioconductor. Ruby and R run in **separate processes**; the **Galaaz bridge**
71
+ sends requests to R and returns results to Ruby. From your point of view you still write
72
+ Ruby: `R.c(...)`, `R.library('ggplot2')`, `~R[:mtcars]`, and dplyr-style chains on R objects.
73
+ You do not need to learn R syntax to get a lot done, though reading R documentation for
74
+ individual packages remains useful.
75
+
76
+ Earlier experiments with Galaaz used Oracle’s **GraalVM** with TruffleRuby and FastR so that
77
+ Ruby and R could share one runtime. That path is no longer the focus: **standard GNU R**
78
+ gives full compatibility with the R package ecosystem (including compiled extensions and
79
+ Bioconductor) while JRuby gives a mature Ruby with **real multithreading** for application
80
+ and I/O code.
81
+
82
+ The bridge handles **communication and typing** between the two worlds; large tables can
83
+ also flow through **Apache Arrow** on the R side when you use the optional helpers described
84
+ later in this manual.
85
+
86
+ Library wrapping is a common way to bring features from one language into another.
87
+ To improve performance, Python often wraps more efficient C libraries. For the
88
+ Python developer, the existence of such C libraries is hidden. The problem with
89
+ library wrapping is that for any new library, there is the need to handcraft a new
90
+ wrapper.
91
+
92
+ Galaaz, instead of wrapping a single C or R library, wraps the whole R language
93
+ in Ruby. Doing so, all thousands of R libraries are available immediately
94
+ to Ruby developers without any new wrapping effort.
95
+
96
+ ## What does Galaaz mean
97
+
98
+ Galaaz is the Portuguese name for "Galahad". From Wikipedia:
99
+
100
+ Sir Galahad (sometimes referred to as Galeas or Galath),
101
+ in Arthurian legend, is a knight of King Arthur's Round Table and one
102
+ of the three achievers of the Holy Grail. He is the illegitimate son
103
+ of Sir Lancelot and Elaine of Corbenic, and is renowned for his
104
+ gallantry and purity as the most perfect of all knights. Emerging quite
105
+ late in the medieval Arthurian tradition, Sir Galahad first appears in the
106
+ Lancelot–Grail cycle, and his story is taken up in later works such as
107
+ the Post-Vulgate Cycle and Sir Thomas Malory's Le Morte d'Arthur.
108
+ His name should not be mistaken with Galehaut, a different knight from
109
+ Arthurian legend.
110
+
111
+ # Command-line tools (`bin/`)
112
+
113
+ The Galaaz repository ships many helpers under **`bin/`**. When working from a **clone**, call
114
+ them as **`bin/<name>`** from the project root (or `./bin/<name>`). If you install the **gem**,
115
+ only a subset is guaranteed on your `PATH` (see the gemspec: **`galaaz`**, **`gstudio`**, **`gknit`**, **`grun`**, **`gknit-draft`**); for development and CI, prefer the **`bin/`** copies so JVM flags and paths stay correct.
116
+
117
+ Below, **current (Galaaz 2.0 + JRuby + GNU R)** means the tool is wired to **`jruby`** and
118
+ **`bin/galaaz_jruby_env.inc.sh`** (or equivalent logic in Ruby via `lib/galaaz_jruby.rb`). **Legacy**
119
+ means the script still targets **GraalVM** polyglot Ruby / FastR-era invocation and is **not**
120
+ expected to work on a typical JRuby-only setup.
121
+
122
+ **Table layout:** names in the first column are **`bin/`** filenames (run as `bin/<name>` from the repo root). Long options and examples sit **outside** the tables so PDF columns stay readable.
123
+
124
+ ```{r bin-tables-helper, echo=FALSE}
125
+ bin_tbl <- function(df) {
126
+ k <- knitr::kable(df, row.names = FALSE, booktabs = TRUE, linesep = "",
127
+ col.names = c("Script", "Role", "2.0?"))
128
+ if (knitr::is_latex_output()) {
129
+ k <- kableExtra::kable_styling(k, font_size = 9, latex_options = "scale_down")
130
+ k <- kableExtra::column_spec(k, 1, width = "2.5cm")
131
+ k <- kableExtra::column_spec(k, 2, width = "9.5cm")
132
+ k <- kableExtra::column_spec(k, 3, width = "2.8cm")
133
+ } else {
134
+ k <- kableExtra::kable_styling(k, bootstrap_options = c("striped", "condensed"), full_width = TRUE)
135
+ }
136
+ k
137
+ }
138
+ ```
139
+
140
+ ```{r bin-tables-bootstrap, echo=FALSE}
141
+ df_boot <- data.frame(
142
+ Script = c("galaaz-bootstrap", "galaaz-jruby", "galaaz_jruby_env.inc.sh", "install-tinytex"),
143
+ Role = c(
144
+ "WSL2 helper: Docker checks; optional TinyTeX or poppler for gKnit PDF.",
145
+ "JRuby with repo lib/ on LOAD_PATH and required JVM flags (e.g. Arrow).",
146
+ "Sourced by bash wrappers; sets GALAAZ_REQUIRED_JRUBY_J_ARGS.",
147
+ "Install TinyTeX for PDF output."
148
+ ),
149
+ X2 = c("Yes*", "Yes", "Yes†", "Yes"),
150
+ stringsAsFactors = FALSE
151
+ )
152
+ bin_tbl(df_boot)
153
+ ```
154
+
155
+ \* Where WSL/Docker apply. **`galaaz-bootstrap` flags:** `--check`, `--apply`, `--runtime` (`docker` \| `local` \| `auto`), `--[no-]prompt-doc-tools`.
156
+
157
+ † Not run directly.
158
+
159
+ **`galaaz-jruby` examples** (from repo root):
160
+
161
+ ```text
162
+ bin/galaaz-jruby my_script.rb
163
+ bin/galaaz-jruby -S rspec
164
+ ```
165
+
166
+ ## Interactive use, examples, and Rake
167
+
168
+ ```{r bin-tables-interactive, echo=FALSE}
169
+ df_ix <- data.frame(
170
+ Script = c("gstudio", "run_example", "galaaz"),
171
+ Role = c(
172
+ "IRB or Pry with Galaaz preloaded (JRuby + JVM flags).",
173
+ "Run one Ruby file using the same JRuby/JVM setup as tests.",
174
+ "Forward arguments to rake (needs rake; usually JRuby)."
175
+ ),
176
+ X2 = c("Yes", "Yes", "Yes"),
177
+ stringsAsFactors = FALSE
178
+ )
179
+ bin_tbl(df_ix)
180
+ ```
181
+
182
+ ## gKnit and document drafts
183
+
184
+ ```{r bin-tables-gknit, echo=FALSE}
185
+ df_gk <- data.frame(
186
+ Script = c("gknit", "gknit-draft", "gknit-draft.rb", "gknit_Rscript"),
187
+ Role = c(
188
+ "Knit .Rmd via JRuby and R Markdown render.",
189
+ "Drafts from rticles-style templates; wrapper still uses legacy polyglot ruby.",
190
+ "Ruby entry: GKnit.draft (use with JRuby + LOAD_PATH).",
191
+ "Polyglot Rscript launcher; hard-coded LOAD_PATH sample."
192
+ ),
193
+ X2 = c("Yes", "Legacy", "JRuby", "No"),
194
+ stringsAsFactors = FALSE
195
+ )
196
+ bin_tbl(df_gk)
197
+ ```
198
+
199
+ **`gknit` CLI** (see `gknit -h`): `--output_format`, `--output_file`, `--output_dir`, `--bridge_timeout_sec`, `--callback_timeout_ms`. If `--output_format` is omitted, the **first** YAML `output:` target wins.
200
+
201
+ Prefer **`galaaz-jruby`** for **`gknit-draft`** workflows until that wrapper matches the **`gknit`** stack.
202
+
203
+ ## Tests
204
+
205
+ ```{r bin-tables-tests, echo=FALSE}
206
+ df_ts <- data.frame(
207
+ Script = c("run_rspec", "run_all_rspec", "run_slow_rspec", "run_old_rspec", "run_rspec_subset"),
208
+ Role = c(
209
+ "Top-level specs/*_spec.rb with spec_helper (see docs/testing.md).",
210
+ "Compile ext/new_bridge; run specs/ and new_bridge_specs/ together.",
211
+ "Suites under slow-specs/ (read script header for spec_helper).",
212
+ "Legacy suites under old_specs/.",
213
+ "Numbered subset 1–18 (Documentation/Spec_Subsets.md)."
214
+ ),
215
+ X2 = c("Yes", "Yes", "Yes", "Yes", "Yes"),
216
+ stringsAsFactors = FALSE
217
+ )
218
+ bin_tbl(df_ts)
219
+ ```
220
+
221
+ ## Other
222
+
223
+ ```{r bin-tables-other, echo=FALSE}
224
+ df_ot <- data.frame(
225
+ Script = c("grun", "gstudio_irb.rb / gstudio_pry.rb"),
226
+ Role = c(
227
+ "Graal-era launcher: polyglot ruby with --jvm. Use galaaz-jruby -S instead.",
228
+ "Loaded by gstudio; not meant to be run standalone."
229
+ ),
230
+ X2 = c("No", "Yes"),
231
+ stringsAsFactors = FALSE
232
+ )
233
+ bin_tbl(df_ot)
234
+ ```
235
+
236
+ For day-to-day **2.0** use, rely on **`bin/galaaz-jruby`**, **`bin/gstudio`**, **`bin/gknit`**, **`bin/run_example`**, **`bin/run_rspec`** / **`bin/run_all_rspec`**, and **`bin/galaaz-bootstrap`** on WSL when using Dockerized R. Treat **`grun`**, **`gknit_Rscript`**, and the polyglot **`ruby`** invocation in **`gknit-draft`** as **legacy** until they are ported to the same JRuby path as **`gknit`**.
35
237
 
36
238
  # System Compatibility
37
239
 
38
- * Oracle Linux 7
39
- * Ubuntu 18.04 LTS
40
- * Ubuntu 16.04 LTS
41
- * Fedora 28
42
- * macOS 10.14 (Mojave)
43
- * macOS 10.13 (High Sierra)
240
+ Typical development and CI targets:
44
241
 
45
- # Dependencies
242
+ * **Linux** — recent Ubuntu LTS or comparable distributions (x86_64).
243
+ * **macOS** — recent releases with JRuby and GNU R available.
244
+ * **Windows** — use **WSL2** (same Linux stack as above); native Windows is not the primary target.
245
+
246
+ The native **gatekeeper** component under `ext/new_bridge` is built with `make` and a C++ toolchain; see the project `README` if compilation fails on your platform.
46
247
 
47
- * TruffleRuby
48
- * FastR
248
+ # Dependencies
49
249
 
250
+ * **JRuby** — Galaaz 2.0 requires JRuby (tested with **10.1.1.0**) and a matching **JDK** (tested with **Java 21**). MRI Ruby is not supported.
251
+ * **GNU R** — `R` and `Rscript` on your `PATH` (tested with **4.3.3**), plus a C++ toolchain (`g++`, `make`) and the **Rcpp** package to compile the gatekeeper.
252
+ * **galaaz gem** — runtime dependency `msgpack` is pulled in by `gem install`.
253
+ * Optional: **Docker** — if you run R in a container (common on WSL2); see bootstrap below.
254
+ * Optional R packages for examples in this manual — e.g. `ggplot2`, `dplyr`, `knitr`, `kableExtra`, `arrow`, Bioconductor tools such as **DESeq2** (installed the usual R way).
50
255
 
51
256
  # Installation
52
257
 
53
- * Install GrallVM (http://www.graalvm.org/)
54
- * Install Ruby (gu install Ruby)
55
- * Install FastR (gu install R)
56
- * Install rake if you want to run the specs and examples (gem install rake)
258
+ The supported install is **`gem install` + compile the gatekeeper**. You do not need a git clone.
259
+
260
+ 1. Install **JRuby**, a compatible **JDK**, and **GNU R** (with `Rscript` and a C++ compiler).
261
+ 2. In R, install **Rcpp**: `install.packages("Rcpp")`.
262
+ 3. Install the gem: `jruby -S gem install galaaz`
263
+ 4. Compile the native gatekeeper from the installed gem:
264
+
265
+ ```
266
+ gem_dir="$(jruby -e "puts Gem::Specification.find_by_name('galaaz').full_gem_path")"
267
+ make -C "${gem_dir}/ext/new_bridge" all
268
+ ```
269
+
270
+ 5. Ensure **`R`** starts GNU R and can install packages (network access to CRAN when you first call `R.install_and_loads`). For **Apache Arrow** on Java 9+, pass `-J--add-opens=java.base/java.nio=ALL-UNNAMED` to JRuby (from a checkout, `bin/galaaz-jruby` does this).
271
+
272
+ For **gKnit**, **knitr**, **rmarkdown**, and LaTeX (PDF output), install the corresponding R packages, **Pandoc**, and a TeX distribution if you need PDF; the repository includes helpers such as **`bin/install-tinytex`** where appropriate.
273
+
274
+ A **table of all `bin/` scripts** (bootstrap, JRuby wrapper, gstudio, gknit, test runners, and which ones are legacy) is in the section **Command-line tools (`bin/`)** earlier in this manual.
275
+
276
+ ### From a repository checkout (contributors)
277
+
278
+ 1. Install **bundler** if needed, then run **`jruby -S bundle install`** in the repository root.
279
+ 2. Build the bridge native code: **`make -C ext/new_bridge all`** (or **`rake compile_gatekeeper`**).
280
+ 3. Run scripts with **`bin/galaaz-jruby`** (sources **`bin/galaaz_jruby_env.inc.sh`** and adds **`-I lib`**).
281
+
282
+ Maintainers can prove a built `.gem` on a throwaway Ubuntu machine (no repo inside the container) with **`./docker/cold-install/run.sh`**.
283
+
284
+ ## Windows + WSL2 (optional: Docker / R in a container)
285
+
286
+ If you run Galaaz on Windows through WSL2 and want containerized R instances,
287
+ Docker Desktop is the supported setup.
288
+
289
+ 1. Install Docker Desktop on Windows:
290
+ - https://www.docker.com/products/docker-desktop/
291
+ 2. Open Docker Desktop and enable WSL integration:
292
+ - Settings > Resources > WSL Integration
293
+ - Enable integration for your target distro
294
+ - Apply & Restart Docker Desktop
295
+ 3. In WSL, run Galaaz bootstrap:
296
+
297
+ > ruby bin/galaaz-bootstrap --apply
298
+ > ruby bin/galaaz-bootstrap --check
299
+
300
+ Expected result:
301
+ - docker CLI available
302
+ - docker compose available
303
+ - docker daemon reachable (`docker info` works)
304
+
305
+ If bootstrap reports daemon is unreachable, check Docker Desktop is running and
306
+ WSL integration is enabled for the distro where Galaaz is installed.
57
307
 
58
308
  # Usage
59
309
 
@@ -82,8 +332,8 @@ Panda, SciPy, SciKit-Learn and a couple more.
82
332
 
83
333
  > galaaz -T
84
334
 
85
- Shows a list with all available executalbe tasks. To execute a task, substitute the
86
- 'rake' word in the list with 'galaaz'. For instance, the following line shows up
335
+ Shows a list with all available executable tasks. To execute a task, substitute the
336
+ 'rake' word in the list with 'galaaz'. For instance, the following line shows up
87
337
  after 'galaaz -T'
88
338
 
89
339
  rake master_list:scatter_plot # scatter_plot from:....
@@ -92,18 +342,948 @@ Panda, SciPy, SciKit-Learn and a couple more.
92
342
 
93
343
  > galaaz master_list:scatter_plot
94
344
 
345
+ # JRuby, multithreading, and the R bridge
346
+
347
+ Galaaz 2.0 runs Ruby on **JRuby**, so your application can use **real parallel threads** for
348
+ I/O-bound work (HTTP clients, database connections, message consumers, and so on). R itself is
349
+ still executed in a **single GNU R process** behind the Galaaz bridge.
350
+
351
+ When several Ruby threads call into R at the same time, the bridge **serializes** those calls:
352
+ each request is matched to a reply using an internal per-call **queue**, so you do not need to
353
+ add your own mutex around every `R.foo` from application threads. (You should still use normal
354
+ Ruby synchronization when **Ruby** data structures are shared between threads—for example, when
355
+ appending rows from each thread into a shared array before sending them to R.)
356
+
357
+ A practical pattern is:
358
+
359
+ 1. Use threads (or a connection pool) to read from **multiple databases or shards** in parallel.
360
+ 2. Merge the rows in Ruby under a `Mutex` if you collect into one structure.
361
+ 3. Hand the merged table to R **once** (for example with `R::Arrow.from_ruby_batches` and dplyr,
362
+ or by building a data frame) so heavy statistics run in R with fewer bridge round-trips.
363
+
364
+ A runnable sketch lives in
365
+ `examples/multithread_shards_to_r/shards_to_r.rb` (simulated shard queries; swap in your DB
366
+ driver). For concurrency tests on the bridge itself, see `specs/bridge_concurrent_spec.rb` and
367
+ `specs/arrow_from_ruby_batches_spec.rb`.
368
+
369
+ ## Long-running R calls and a completion block
370
+
371
+ For R work that can take a long time, the bridge can avoid a Ruby-side **wait timeout** by
372
+ scheduling the call and resuming in a **block** when the `RET` arrives.
373
+
374
+ - **`R.eval_r_async(code, timeout: nil) { |result| ... }`** — string eval; on success, `result.value`
375
+ is the same formatted string as **`R.eval_r`** (use `timeout: nil` for no Ruby-side limit).
376
+ - **`R::Async.<rname>(...) { |result| ... }`** — same dispatch as **`R.<rname>(...)`**, but async;
377
+ on success, `result.value` is an **`R::Object`** (or unboxed Ruby value / Symbol), like synchronous
378
+ **`R.<rname>`**. Optional keyword **`timeout:`** applies a Ruby-side wait limit (completion receives
379
+ **`NewBridge::SessionClient::TimeoutError`** if R is too slow).
380
+
381
+ **Important:** **`R.foo(...) { |x| }`** is already used for dplyr-style scopes (`R::Support.new_scope`),
382
+ so async R calls must use **`R::Async`** or **`R.eval_r_async`**, not a bare **`R.foo` with a block.**
383
+
384
+ `NewBridge::EvalResult` exposes **`#ok?`**, **`#value`**, and **`#error`**. The completion block runs on a
385
+ **background thread** (not the bridge reader thread).
386
+
387
+ The example below is **plain Ruby** (no Rails). The R snippet sleeps (standing in for heavy work) and then
388
+ returns an integer so the success branch shows a **non-nil** value. (`Sys.sleep` alone returns **NULL** in R;
389
+ on success **`result.value`** is then **`nil`** in Ruby—that is expected, not a bridge error.)
390
+
391
+ ```{ruby long_r_completion_block}
392
+ require 'thread'
393
+
394
+ completion = Queue.new
395
+
396
+ R.eval_r_async('({ Sys.sleep(0.3); 42L })', timeout: nil) do |result|
397
+ if result.ok?
398
+ puts "[completion] R finished; eval_r-style value: #{result.value.inspect}"
399
+ else
400
+ puts "[completion] R/bridge error: #{result.error.class}: #{result.error.message}"
401
+ end
402
+ completion.push(:done)
403
+ end
404
+
405
+ 3.times do |i|
406
+ puts "[main] other Ruby work step #{i + 1}"
407
+ sleep 0.05
408
+ end
409
+
410
+ completion.pop
411
+ puts "[main] R completion has run; exiting."
412
+ ```
413
+
414
+ In a **web application**, the HTTP response usually ends before R finishes, so you would not
415
+ `Queue#pop` in the controller; you would persist an identifier, let the completion block write
416
+ the outcome to storage, and notify the client (poll, WebSocket, Turbo Stream, etc.). The plain
417
+ Ruby pattern above is only to show **when** the result exists (inside the block, or after data
418
+ written there is observed elsewhere). Runnable specs live in **`new_bridge_specs/eval_r_async_spec.rb`**.
419
+
420
+ ## Galaaz + Rails (JRuby) integration baseline
421
+
422
+ This section documents the baseline we used to create a working Rails app with Galaaz in WSL.
423
+ The goals were:
424
+
425
+ 1. Rails boots under **JRuby**.
426
+ 2. Galaaz is loaded from a local checkout (before publishing to RubyGems).
427
+ 3. A request path can execute **`R.eval(...)`** and return a result.
428
+
429
+ ### 1) Create the app with JRuby-friendly options
430
+
431
+ Rails defaults can pull gems that are not ideal on JRuby-first setups (for example sqlite native
432
+ extension paths and deployment extras). A minimal app avoids early friction:
433
+
434
+ ```bash
435
+ cd /home/rbotafogo/desenv_linux
436
+ jruby -S rails new hedi --skip-git --minimal --skip-kamal --skip-solid --skip-active-record
437
+ ```
438
+
439
+ Then install gems:
440
+
441
+ ```bash
442
+ cd /home/rbotafogo/desenv_linux/hedi
443
+ jruby -S bundle install
444
+ ```
445
+
446
+ ### 2) Use Galaaz as a local path gem
447
+
448
+ For local development we keep a stable path:
449
+
450
+ - `~/gems/galaaz` -> symlink to your Galaaz checkout
451
+ - optional built gem archive in `~/gems/pkg/`
452
+
453
+ In Rails `Gemfile`:
454
+
455
+ ```ruby
456
+ gem "galaaz", path: "/home/rbotafogo/gems/galaaz", require: false
457
+ ```
458
+
459
+ And load after Rails boot in `config/application.rb`:
460
+
461
+ ```ruby
462
+ config.after_initialize { require "galaaz" }
463
+ ```
464
+
465
+ Why `require: false` + `after_initialize`? In this integration, loading Galaaz too early via
466
+ `Bundler.require` triggered Rails/JRuby initialization failures.
467
+
468
+ ### 3) Simple request-path smoke test
469
+
470
+ A direct smoke test from Rails runner:
471
+
472
+ ```bash
473
+ cd /home/rbotafogo/desenv_linux/hedi
474
+ jruby -S bundle exec rails runner "puts R.eval('sum(c(1,2,3,4,5))').inspect"
475
+ ```
476
+
477
+ Expected output:
478
+
479
+ ```text
480
+ 15.0
481
+ ```
482
+
483
+ ### 4) HTTP endpoint pattern
484
+
485
+ For this baseline, a small Rack endpoint was the most stable first step to prove request-time R
486
+ evaluation. (A full ActionController stack can be enabled later as the app evolves.)
487
+
488
+ Minimal pattern:
489
+
490
+ 1. Define a Rack app class under `lib/` that runs `R.eval(...)` and returns HTML/JSON.
491
+ 2. Point a route to that Rack app (`root to: MyRackApp`).
492
+ 3. Verify with browser/curl.
493
+
494
+ ### 5) Running from WSL and opening from Windows
495
+
496
+ Recommended bind:
497
+
498
+ ```bash
499
+ jruby -S bundle exec rails server -b 0.0.0.0 -p 3000
500
+ ```
501
+
502
+ Then open from Windows:
503
+
504
+ - `http://localhost:3000` (usually works with WSL localhost forwarding), or
505
+ - `http://<wsl-ip>:3000` if needed.
506
+
507
+ In development, if Host Authorization blocks requests with unexpected Host headers, use:
508
+
509
+ ```ruby
510
+ # config/environments/development.rb
511
+ config.hosts.clear
512
+ ```
513
+
514
+ ### 6) Troubleshooting checklist
515
+
516
+ - `Could not find ... in locally installed gems`:
517
+ run `jruby -S bundle install` in the Rails app directory.
518
+ - Stale PID after crash:
519
+ remove `tmp/pids/server.pid`.
520
+ - Local Galaaz path changed:
521
+ verify `Gemfile` path target exists and rerun bundler.
522
+ - R runtime issues:
523
+ confirm GNU R is installed and on `PATH` in the same shell where Rails runs.
524
+
525
+ As new Rails features are added (controllers, jobs, websockets, background rendering, plot
526
+ generation), extend this section with concrete, runnable snippets and the associated operational
527
+ checks.
528
+
529
+ # Accessing R from Ruby
530
+
531
+ One of the nice aspects of Galaaz is that variables and functions defined in R can
532
+ be easily accessed from Ruby. For instance, to access the `mtcars` data frame from R
533
+ in Ruby, we use the symbol `:mtcars` preceded by the `~` operator: `~R[:mtcars]` retrieves the
534
+ value of the `mtcars` object in R.
535
+
536
+ ```{ruby access_r}
537
+ puts ~R[:mtcars]
538
+ ```
539
+
540
+ ## Scoped symbols and lexical scoping
541
+
542
+ Galaaz 2.0 uses **scoped symbols** by default. The canonical style is `R[:name]`:
543
+
544
+ - `~R[:mtcars]` fetches an R object by name.
545
+ - `R[:a] + R[:b]` builds an expression.
546
+ - `R[:year].up_to(R[:day])` builds range expressions.
547
+
548
+ If you prefer the terse `:x` syntax, you can opt in with lexical scoping using a Ruby refinement:
549
+
550
+ ```ruby
551
+ module MyScript
552
+ using Galaaz::SymbolDSL
553
+
554
+ def self.run
555
+ expr = :a + :b
556
+ puts expr
557
+ puts ~:mtcars
558
+ end
559
+ end
560
+ ```
561
+
562
+ `using Galaaz::SymbolDSL` is **lexically scoped**: only code in that module/file scope gets `:x` DSL behavior.
563
+ Outside that scope, plain Ruby `Symbol` behavior is unchanged.
564
+
565
+ To access an R function from Ruby, the R function needs to be preceded by `R.` scoping.
566
+ Below we see an example of creating a R::Vector by calling the 'c' R function
567
+
568
+ ```{ruby call_r_func}
569
+ puts vec = R.c(1.0, 2.0, 3.0, 4.0)
570
+ ```
571
+ Note that 'vec' is an object of type R::Vector:
572
+
573
+ ```{ruby r_object}
574
+ puts vec.class
575
+ ```
576
+ Every object created by a call to an R function will be of a type that inherits from
577
+ R::Object. In R, there is also a function 'class'. In order to access that function we
578
+ can call method 'rclass' in the R::Object:
579
+
580
+ ```{ruby rclass}
581
+ puts vec.rclass
582
+ ```
583
+ When working with R::Object(s), it is possible to use the '.' operator to pipe operations.
584
+ When using '.', the object to which the '.' is applied becomes the first argument of the
585
+ corresponding R function. For instance, function 'c' in R, can be used to concatenate
586
+ two vectors or more vectors (in R, there are no scalar values, scalars are converted to
587
+ vectors of size 1. Within Galaaz, scalar parameter is converted to a size one vector):
588
+
589
+ ```{ruby concat}
590
+ puts R.c(vec, 10, 20, 30)
591
+ ```
592
+ The call above to the 'c' function can also be done using '.' notation:
593
+
594
+ ```{ruby concat_with_dot}
595
+ puts vec.c(10, 20, 30)
596
+ ```
597
+ We will talk about vector indexing in a later section. But notice here that indexing
598
+ an R::Vector will return another R::Vector:
599
+
600
+ ```{ruby indexing}
601
+ puts vec[1]
602
+ ```
603
+ Sometimes we want to index an R::Object and get back a Ruby object that is not wrapped
604
+ in an R::Object, but the native Ruby object. For this, we can index the R object with
605
+ the '>>' operator:
606
+
607
+ ```{ruby native_value}
608
+ puts vec >> 0
609
+ puts vec >> 2
610
+ ```
611
+
612
+ It is also possible to call an R function with named arguments, by creating the function
613
+ in Galaaz with named parameters. For instance, here is an example of creating a 'list'
614
+ with named elements:
615
+
616
+ ```{ruby named_parameters}
617
+ puts R.list(first_name: "Rodrigo", last_name: "Botafogo")
618
+ ```
619
+
620
+ Many R functions receive another function as argument. For instance, method 'map' applies
621
+ a function to every element of a vector. With Galaaz, it is possible to pass a Proc,
622
+ Method or Lambda in place of the expected R function. In this next example, we will
623
+ add 2 to every element of our previously created vector:
624
+
625
+ ```{ruby proc_as_param}
626
+ puts vec.map { |x| x + 2 }
627
+ ```
628
+
95
629
  # gKnitting a Document
96
630
 
97
- This manual has been formatted usign gKnit. gKnit uses Knitr and R markdown to knit
98
- a document in Ruby or R and output it in any of the available formats for R markdown.
99
- gKnit runs atop of GraalVM, and Galaaz. In gKnit, Ruby variables are persisted between
631
+ This manual has been formatted using gKnit. gKnit uses knitr and R Markdown to knit
632
+ a document in Ruby or R and output it in any of the available formats for R Markdown.
633
+ gKnit runs with **JRuby**, **GNU R**, and Galaaz. In gKnit, Ruby variables are persisted between
100
634
  chunks, making it an ideal solution for literate programming. Also, since it is based
101
- on Galaaz, Ruby chunks can have access to R variables and Polyglot Programming with
102
- Ruby and R is quite natural.
635
+ on Galaaz, Ruby chunks can have access to R variables and combining Ruby with R in one
636
+ document is natural.
637
+
638
+ The idea of "literate programming" was first introduced by Donald Knuth in the
639
+ 1980's [@Knuth:literate_programming].
640
+ The main intention of this approach was to develop software interspersing macro snippets,
641
+ traditional source code, and a natural language such as English in a document
642
+ that could be compiled into
643
+ executable code and at the same time easily read by a human developer. According to Knuth
644
+ "The practitioner of
645
+ literate programming can be regarded as an essayist, whose main concern is with exposition
646
+ and excellence of style."
647
+
648
+ The idea of literate programming evolved into the idea of reproducible research, in which
649
+ all the data, software code, documentation, graphics etc. needed to reproduce the research
650
+ and its reports could be included in a
651
+ single document or set of documents that when distributed to peers could be rerun generating
652
+ the same output and reports.
653
+
654
+ The R community has put a great deal of effort in reproducible research. In 2002, Sweave was
655
+ introduced and it allowed mixing R code with LaTeX, generating high-quality PDF documents. A
656
+ Sweave document could include code, the results of executing the code, graphics and text
657
+ such that it contained the whole narrative to reproduce the research. In
658
+ 2012, Knitr, developed by Yihui Xie from RStudio was released to replace Sweave and to
659
+ consolidate in one single package the many extensions and add-on packages that
660
+ were necessary for Sweave.
661
+
662
+ With Knitr, __R markdown__ was also developed, an extension to the
663
+ Markdown format. With __R markdown__ and Knitr it is possible to generate reports in a multitude
664
+ of formats such as HTML, Markdown, LaTeX, PDF, DVI, etc. __R markdown__ also allows the use of
665
+ multiple programming languages such as R, Ruby, Python, etc. in the same document.
666
+
667
+ In __R markdown__, text is interspersed with
668
+ code chunks that can be executed and both the code and its results can become
669
+ part of the final report. Although __R markdown__ allows multiple programming languages in the
670
+ same document, only R and Python (with
671
+ the reticulate package) can persist variables between chunks. For other languages, such as
672
+ Ruby, every chunk will start a new process and thus all data is lost between chunks, unless it
673
+ is somehow stored in a data file that is read by the next chunk.
674
+
675
+ Being able to persist data
676
+ between chunks is critical for literate programming otherwise the flow of the narrative is lost
677
+ by all the effort of having to save data and then reload it. Although this might, at first, seem like
678
+ a small nuisance, not being able to persist data between chunks is a major issue. For example, let's
679
+ take a look at the following simple example in which we want to show how to create a list and the
680
+ use it. Let's first assume that data cannot be persisted between chunks. In the next chunk we
681
+ create a list, then we would need to save it to file, but to save it, we need somehow to marshal the
682
+ data into a binary format:
683
+
684
+ ```{ruby no_persistence}
685
+ lst = R.list(a: 1, b: 2, c: 3)
686
+ lst.saveRDS("lst.rds")
687
+ ```
688
+ then, on the next chunk, where variable 'lst' is used, we need to read back it's value
689
+
690
+ ```{ruby load_persisted_data}
691
+ lst = R.readRDS("lst.rds")
692
+ puts lst
693
+ ```
694
+
695
+ Now, any single code has dozens of variables that we might want to use and reuse between chunks.
696
+ Clearly, such an approach becomes quickly unmanageable. Probably, because of
697
+ this problem, it is very rare to see any __R markdown__ document in the Ruby community.
698
+
699
+ When variables can be used across chunks, then no overhead is needed:
700
+
701
+ ```{ruby persistence}
702
+ lst = R.list(a: 1, b: 2, c: 3)
703
+ # any other code can be added here
704
+ ```
705
+
706
+ ```{ruby use_var}
707
+ puts lst
708
+ ```
709
+
710
+ In the Python community, the same effort to have code and text in an integrated environment
711
+ started around the first decade of the 2000s. In 2006 IPython 0.7.2 was released. In 2014,
712
+ Fernando Pérez spun off the Jupyter project from IPython, creating a web-based interactive
713
+ computation environment. Jupyter can now be used with many languages, including Ruby with the
714
+ iruby gem (https://github.com/SciRuby/iruby). In order to have multiple languages in a Jupyter
715
+ notebook the SoS kernel was developed (https://vatlab.github.io/sos-docs/).
716
+
717
+ ## gKnit and __R markdown__
718
+
719
+ gKnit is based on knitr and __R markdown__ and can knit a document
720
+ written both in Ruby and/or R and output it in any of the available formats of __R markdown__. gKnit
721
+ allows ruby developers to do literate programming and reproducible research by allowing them to
722
+ have in a single document, text and code.
723
+
724
+ In gKnit, Ruby variables are persisted between
725
+ chunks, making it an ideal solution for literate programming in this language. Also,
726
+ since it is based on Galaaz, Ruby chunks can access R variables (`~R[:name]`, `R.*`) through the
727
+ **Galaaz bridge** while knitr drives **GNU R**—no GraalVM polyglot runtime is required.
728
+
729
+ This is not a blog post on __R markdown__, and the interested user is directed to the following links
730
+ for detailed information on its capabilities and use.
731
+
732
+ * https://rmarkdown.rstudio.com/ or
733
+ * https://bookdown.org/yihui/rmarkdown/
734
+
735
+ In this post, we will describe just the main aspects of __R markdown__, so the user can start
736
+ gKnitting Ruby and R documents quickly.
737
+
738
+ ## The Yaml header
739
+
740
+ An __R markdown__ document should start with a Yaml header and be stored in a file with
741
+ '.Rmd' extension. This document has the following header for gKnitting an HTML document.
742
+
743
+ ```
744
+ ---
745
+ title: "How to do reproducible research in Ruby with gKnit"
746
+ author:
747
+ - "Rodrigo Botafogo"
748
+ - "Daniel Mossé - University of Pittsburgh"
749
+ tags: [Tech, Data Science, Ruby, R, JRuby, Galaaz]
750
+ date: "20/02/2019"
751
+ output:
752
+ html_document:
753
+ self_contained: true
754
+ keep_md: true
755
+ pdf_document:
756
+ includes:
757
+ in_header: ["../../sty/galaaz.sty"]
758
+ number_sections: yes
759
+ ---
760
+ ```
761
+
762
+ For more information on the options in the Yaml header, [check here](https://bookdown.org/yihui/rmarkdown/html-document.html).
763
+
764
+ ## Choosing the output format when calling gknit
765
+
766
+ Yes: you can select the render target on the **command line**. **`bin/gknit`** (or **`gknit`** on your `PATH`) forwards options to **`rmarkdown::render`** via **`R::Rmarkdown.render`**.
767
+
768
+ * **`--output_format FORMAT`** — name of the format, as in the YAML `output:` block. Examples:
769
+ * **`html_document`** — HTML (often the default you list first under `output:`).
770
+ * **`pdf_document`** — PDF (you need a working LaTeX setup, e.g. TinyTeX; see **`bin/install-tinytex`**).
771
+ * **`md_document`**, **`github_document`**, or any other format defined in your YAML.
772
+ * **`all`** — render **every** format declared under `output:` in the document (same idea as in R Markdown).
773
+
774
+ If you **omit** **`--output_format`**, gknit passes **`NULL`** for the format argument. In that case **rmarkdown** uses the **first** format listed under **`output:`** in the YAML (and if none is specified there, behavior follows the usual rmarkdown defaults, typically HTML).
775
+
776
+ Other useful flags:
777
+
778
+ * **`--output_file NAME`** — output file name (optional path; see also **`--output_dir`**).
779
+ * **`--output_dir DIR`** — directory for the rendered file (created if missing).
780
+ * **`--bridge_timeout_sec`** / **`--callback_timeout_ms`** — longer R or install steps (see elsewhere in this manual).
781
+
782
+ Examples (run from the directory where paths make sense, or use absolute paths):
783
+
784
+ ```text
785
+ bin/gknit blogs/manual/manual.Rmd
786
+ bin/gknit --output_format html_document blogs/manual/manual.Rmd
787
+ bin/gknit --output_format pdf_document blogs/manual/manual.Rmd
788
+ bin/gknit --output_format all blogs/manual/manual.Rmd
789
+ ```
790
+
791
+ Use **`gknit -h`** for the full option list.
792
+
793
+ ## __R Markdown__ formatting
794
+
795
+ Document formatting can be done with simple markups such as:
796
+
797
+ ## Headers
798
+
799
+ ```
800
+ # Header 1
103
801
 
104
- [gknit is described in more details here](https://towardsdatascience.com/how-to-do-reproducible-research-in-ruby-with-gknit-c26d2684d64e)
802
+ ## Header 2
105
803
 
106
- # Vector
804
+ ### Header 3
805
+
806
+ ```
807
+
808
+ ## Lists
809
+
810
+ ```
811
+ Unordered lists:
812
+
813
+ * Item 1
814
+ * Item 2
815
+ + Item 2a
816
+ + Item 2b
817
+ ```
818
+
819
+ ```
820
+ Ordered Lists
821
+
822
+ 1. Item 1
823
+ 2. Item 2
824
+ 3. Item 3
825
+ + Item 3a
826
+ + Item 3b
827
+ ```
828
+
829
+ For more R markdown formatting go to https://rmarkdown.rstudio.com/authoring_basics.html.
830
+
831
+ ## R chunks
832
+
833
+ Running and executing Ruby and R code is actually what really interests us is this blog.
834
+ Inserting a code chunk is done by adding code in a block delimited by three back ticks
835
+ followed by an open
836
+ curly brace ('{') followed with the engine name (r, ruby, rb, include, ...), an
837
+ any optional chunk_label and options, as shown below:
838
+
839
+ ````
840
+ ```{engine_name [chunk_label], [chunk_options]}`r ''`
841
+ ```
842
+ ````
843
+
844
+ for instance, let's add an R chunk to the document labeled 'first_r_chunk'. This is
845
+ a very simple code just to create a variable and print it out, as follows:
846
+
847
+ ````
848
+ ```{r first_r_chunk}`r ''`
849
+ vec <- c(1, 2, 3)
850
+ print(vec)
851
+ ```
852
+ ````
853
+
854
+ If this block is added to an __R markdown__ document and gKnitted the result will be:
855
+
856
+ ```{r first_r_chunk}
857
+ vec <- c(1, 2, 3)
858
+ print(vec)
859
+ ```
860
+
861
+ Now let's say that we want to do some analysis in the code, but just print the result and not the
862
+ code itself. For this, we need to add the option 'echo = FALSE'.
863
+
864
+ ````
865
+ ```{r second_r_chunk, echo = FALSE}`r ''`
866
+ vec2 <- c(10, 20, 30)
867
+ vec3 <- vec * vec2
868
+ print(vec3)
869
+ ```
870
+ ````
871
+ Here is how this block will show up in the document. Observe that the code is not shown
872
+ and we only see the execution result in a white box
873
+
874
+ ```{r second_r_chunk, echo = FALSE}
875
+ vec2 <- c(10, 20, 30)
876
+ vec3 <- vec * vec2
877
+ print(vec3)
878
+ ```
879
+
880
+ A description of the available chunk options can be found in https://yihui.name/knitr/.
881
+
882
+ Let's add another R chunk with a function definition. In this example, a vector
883
+ 'r_vec' is created and
884
+ a new function 'reduce_sum' is defined. The chunk specification is
885
+
886
+ ````
887
+ ```{r data_creation}`r ''`
888
+ r_vec <- c(1, 2, 3, 4, 5)
889
+
890
+ reduce_sum <- function(...) {
891
+ Reduce(sum, as.list(...))
892
+ }
893
+ ```
894
+ ````
895
+
896
+ and this is how it will look like once executed. From now on, to be concise in the
897
+ presentation we will not show chunk definitions any longer.
898
+
899
+
900
+ ```{r data_creation}
901
+ r_vec <- c(1, 2, 3, 4, 5)
902
+
903
+ reduce_sum <- function(...) {
904
+ Reduce(sum, as.list(...))
905
+ }
906
+ ```
907
+
908
+ We can, possibly in another chunk, access the vector and call the function as follows:
909
+
910
+ ```{r using_previous}
911
+ print(r_vec)
912
+ print(reduce_sum(r_vec))
913
+ ```
914
+ ## R Graphics with ggplot
915
+
916
+ In the following chunk, we create a bubble chart in R using ggplot and include it in
917
+ this document. Note that there is no directive in the code to include the image, this
918
+ occurs automatically. The 'mpg' dataframe is natively available to R and to Galaaz as
919
+ well.
920
+
921
+ For the reader not knowledgeable of ggplot, ggplot is a graphics library based on "the
922
+ grammar of graphics" [@Wilkinson:grammar_of_graphics]. The idea of the grammar of graphics
923
+ is to build a graphics by adding layers to the plot. More information can be found in
924
+ https://towardsdatascience.com/a-comprehensive-guide-to-the-grammar-of-graphics-for-effective-visualization-of-multi-dimensional-1f92b4ed4149.
925
+
926
+ In the plot below the 'mpg' dataset from base R is used. "The data concerns city-cycle fuel
927
+ consumption in miles per gallon, to be predicted in terms of 3 multivalued discrete and 5
928
+ continuous attributes." (Quinlan, 1993)
929
+
930
+ First, the 'mpg' dataset if filtered to extract only cars from the following manumactures: Audi, Ford,
931
+ Honda, and Hyundai and stored in the 'mpg_select' variable. Then, the selected dataframe is passed
932
+ to the ggplot function specifying in the aesthetic method (aes) that 'displacement' (disp) should
933
+ be plotted in the 'x' axis and 'city mileage' should be on the 'y' axis. In the 'labs' layer we
934
+ pass the 'title' and 'subtitle' for the plot. To the basic plot 'g', geom\_jitter is added, that
935
+ plots cars from the same manufactures with the same color (col=manufactures) and the size of the
936
+ car point equal its high way consumption (size = hwy). Finally, a last layer is plotter containing
937
+ a linear regression line (method = "lm") for every manufacturer.
938
+
939
+ ```{r bubble, dev='png'}
940
+ # load package and data
941
+ library(ggplot2)
942
+ data(mpg, package="ggplot2")
943
+
944
+ mpg_select <- mpg[mpg$manufacturer %in% c("audi", "ford", "honda", "hyundai"), ]
945
+
946
+ # Scatterplot
947
+ theme_set(theme_bw()) # pre-set the bw theme.
948
+ g <- ggplot(mpg_select, aes(displ, cty)) +
949
+ labs(subtitle="mpg: Displacement vs City Mileage",
950
+ title="Bubble chart")
951
+
952
+ g + geom_jitter(aes(col=manufacturer, size=hwy)) +
953
+ geom_smooth(aes(col=manufacturer), method="lm", se=F)
954
+ ```
955
+
956
+ ## Ruby chunks
957
+
958
+ Including a Ruby chunk is just as easy as including an R chunk in the document: just
959
+ change the name of the engine to 'ruby'. It is also possible to pass chunk options
960
+ to the Ruby engine; however, this version does not accept all the options that are
961
+ available to R chunks. Future versions will add those options.
962
+
963
+ ````
964
+ ```{ruby first_ruby_chunk}`r ''`
965
+ ```
966
+ ````
967
+
968
+ In this example, the ruby chunk is called 'first_ruby_chunk'. One important
969
+ aspect of chunk labels is that they cannot be duplicated. If a chunk label is
970
+ duplicated, gKnit will stop with an error.
971
+
972
+ In the following chunk, variable 'a', 'b' and 'c' are standard Ruby variables
973
+ and 'vec' and 'vec2' are two vectors created by calling the 'c' method on the
974
+ R module.
975
+
976
+ In Galaaz, the R module allows us to access R functions transparently. The 'c'
977
+ function in R, is a function that concatenates its arguments making a vector.
978
+
979
+ It
980
+ should be clear that there is no requirement in gknit to call or use any R
981
+ functions. gKnit will knit standard Ruby code, or even general text without
982
+ any code.
983
+
984
+ ```{ruby split_data}
985
+ a = [1, 2, 3]
986
+ b = "US$ 250.000"
987
+ c = "The 'outputs' function"
988
+
989
+ vec = R.c(1, 2, 3)
990
+ vec2 = R.c(10, 20, 30)
991
+ ```
992
+
993
+ In the next block, variables 'a', 'vec' and 'vec2' are used and printed.
994
+
995
+ ```{ruby split2}
996
+ puts a
997
+ puts vec * vec2
998
+ ```
999
+
1000
+ Note that 'a' is a standard Ruby Array and 'vec' and 'vec2' are vectors that behave accordingly,
1001
+ where multiplication works as expected.
1002
+
1003
+ ## Inline Ruby code
1004
+
1005
+ When using a Ruby chunk, the code and the output are formatted in blocks as seen above.
1006
+ This formatting is not always desired. Sometimes, we want to have the results of the
1007
+ Ruby evaluation included in the middle of a phrase. gKnit allows adding inline Ruby code
1008
+ with the 'rb' engine. The following chunk specification will
1009
+ create and inline Ruby text:
1010
+
1011
+ ````
1012
+ This is some text with inline Ruby accessing variable 'b' which has value:
1013
+ ```{rb puts "```{rb puts b}\n```"}
1014
+ ```
1015
+ and is followed by some other text!
1016
+ ````
1017
+
1018
+ <div style="margin-bottom:30px;">
1019
+ </div>
1020
+
1021
+ This is some text with inline Ruby accessing variable 'b' which has value:
1022
+ ```{rb puts b}
1023
+ ```
1024
+ and is followed by some other text!
1025
+
1026
+ <div style="margin-bottom:30px;">
1027
+ </div>
1028
+
1029
+ Note that it is important not to add any new line before of after the code
1030
+ block if we want everything to be in only one line, resulting in the following sentence
1031
+ with inline Ruby code.
1032
+
1033
+
1034
+ ```{ruby heading, echo = FALSE}
1035
+ outputs "### #{c}"
1036
+ ```
1037
+
1038
+ He have previously used the standard 'puts' method in Ruby chunks in order produce
1039
+ output. The result of a 'puts', as seen in all previous chunks that use it, is formatted
1040
+ inside a white box that
1041
+ follows the code block. Many times however, we would like to do some processing in the
1042
+ Ruby chunk and have the result of this processing generate and output that is
1043
+ "included" in the document as if we had typed it in __R markdown__ document.
1044
+
1045
+ For example, suppose we want to create a new heading in our document, but the heading
1046
+ phrase is the result of some code processing: maybe it's the first line of a file we are
1047
+ going to read. Method 'outputs' adds its output as if typed in the __R markdown__ document.
1048
+
1049
+ Take now a look at variable 'c' (it was defined in a previous block above) as
1050
+ 'c = "The 'outputs' function". "The 'outputs' function" is actually the name of this
1051
+ section and it was created using the 'outputs' function inside a Ruby chunk.
1052
+
1053
+ The ruby chunk to generate this heading is:
1054
+
1055
+ ````
1056
+ ```{ruby heading}`r ''`
1057
+ outputs "### #{c}"
1058
+ ```
1059
+ ````
1060
+
1061
+ The three '###' is the way we add a Heading 3 in __R markdown__.
1062
+
1063
+
1064
+ ### HTML Output from Ruby Chunks
1065
+
1066
+ We've just seen the use of method 'outputs' to add text to the the __R markdown__
1067
+ document. This technique can also be used to add HTML code to the document. In
1068
+ __R markdown__, any html code typed directly in the document will be properly rendered.
1069
+ Here, for instance, is a table definition in HTML and its output in the document:
1070
+
1071
+ ```
1072
+ <table style="width:100%">
1073
+ <tr>
1074
+ <th>Firstname</th>
1075
+ <th>Lastname</th>
1076
+ <th>Age</th>
1077
+ </tr>
1078
+ <tr>
1079
+ <td>Jill</td>
1080
+ <td>Smith</td>
1081
+ <td>50</td>
1082
+ </tr>
1083
+ <tr>
1084
+ <td>Eve</td>
1085
+ <td>Jackson</td>
1086
+ <td>94</td>
1087
+ </tr>
1088
+ </table>
1089
+ ```
1090
+ <div style="margin-bottom:30px;">
1091
+ </div>
1092
+
1093
+ <table style="width:100%">
1094
+ <tr>
1095
+ <th>Firstname</th>
1096
+ <th>Lastname</th>
1097
+ <th>Age</th>
1098
+ </tr>
1099
+ <tr>
1100
+ <td>Jill</td>
1101
+ <td>Smith</td>
1102
+ <td>50</td>
1103
+ </tr>
1104
+ <tr>
1105
+ <td>Eve</td>
1106
+ <td>Jackson</td>
1107
+ <td>94</td>
1108
+ </tr>
1109
+ </table>
1110
+
1111
+ <div style="margin-bottom:30px;">
1112
+ </div>
1113
+
1114
+ But manually creating HTML output is not always easy or desirable, specially
1115
+ if we intend the document to be rendered in other formats, for example, as LaTeX.
1116
+ Also, The above
1117
+ table looks ugly. The 'kableExtra' library is a great library for
1118
+ creating beautiful tables. Take a look at https://cran.r-project.org/web/packages/kableExtra/vignettes/awesome_table_in_html.html
1119
+
1120
+ In the next chunk, we output the 'mtcars' dataframe from R in a nicely formatted
1121
+ table. Note that we retrieve the mtcars dataframe by using '~R[:mtcars]'.
1122
+
1123
+ ```{ruby nice_table}
1124
+ R.install_and_loads('kableExtra')
1125
+ outputs (~R[:mtcars]).kable.kable_styling
1126
+ ```
1127
+
1128
+ ## Including Ruby files in a chunk
1129
+
1130
+ R is a language that was created to be easy and fast for statisticians to use. As far
1131
+ as I know, it was not a
1132
+ language to be used for developing large systems. Of course, there are large systems and
1133
+ libraries in R, but the focus of the language is for developing statistical models and
1134
+ distribute that to peers.
1135
+
1136
+ Ruby on the other hand, is a language for large software development. Systems written in
1137
+ Ruby will have dozens, hundreds or even thousands of files. To document a
1138
+ large system with literate programming, we cannot expect the developer to add all the
1139
+ files in a single '.Rmd' file. gKnit provides the 'include' chunk engine to include
1140
+ a Ruby file as if it had being typed in the '.Rmd' file.
1141
+
1142
+ To include a file, the following chunk should be created, where <filename> is the name of
1143
+ the file to be included and where the extension, if it is '.rb', does not need to be added.
1144
+ If the 'relative' option is not included, then it is treated as TRUE. When 'relative' is
1145
+ true, ruby's 'require\_relative' semantics is used to load the file, when false, Ruby's
1146
+ \$LOAD_PATH is searched to find the file and it is 'require'd.
1147
+
1148
+ ````
1149
+ ```{include <filename>, relative = <TRUE/FALSE>}`r ''`
1150
+ ```
1151
+ ````
1152
+
1153
+ Below we include file 'model.rb', which is in the same directory of this blog.
1154
+ This code uses R 'caret' package to split a dataset in a train and test sets.
1155
+ The 'caret' package is a very important a useful package for doing Data Analysis,
1156
+ it has hundreds of functions for all steps of the Data Analysis workflow. To
1157
+ use 'caret' just to split a dataset is like using the proverbial cannon to
1158
+ kill the fly. We use it here only to show that integrating Ruby and R and
1159
+ using even a very complex package as 'caret' is trivial with Galaaz.
1160
+
1161
+ A word of advice: the 'caret' package has lots of dependencies and installing
1162
+ it in a Linux system is a time consuming operation. Method 'R.install_and_loads'
1163
+ will install the package if it is not already installed and can take a while.
1164
+
1165
+ ````
1166
+ ```{include model}`r ''`
1167
+ ```
1168
+ ````
1169
+
1170
+ ```{include model}
1171
+ ```
1172
+
1173
+ ```{ruby model_partition}
1174
+ mtcars = ~R[:mtcars]
1175
+ model = Model.new(mtcars, percent_train: 0.8)
1176
+ model.partition(:mpg)
1177
+ puts model.train.head
1178
+ puts model.test.head
1179
+ ```
1180
+
1181
+ ## Documenting Gems
1182
+
1183
+ gKnit also allows developers to document and load files that are not in the same directory
1184
+ of the '.Rmd' file.
1185
+
1186
+ Here is an example of loading Ruby’s standard library file `find.rb`. In this example, relative
1187
+ is set to FALSE, so Ruby will look for the file in its `$LOAD_PATH`, and the user does not
1188
+ need to know its directory on disk.
1189
+
1190
+ ````
1191
+ ```{include find, relative = FALSE}`r ''`
1192
+ ```
1193
+ ````
1194
+
1195
+ ```{include find, relative = FALSE}
1196
+ ```
1197
+
1198
+ ## Converting to PDF
1199
+
1200
+ One of the beauties of knitr is that the same input can be converted to many different outputs.
1201
+ One very useful format, is, of course, PDF. In order to converted an __R markdown__ file to PDF
1202
+ it is necessary to have LaTeX installed on the system. We will not explain here how to
1203
+ install LaTeX as there are plenty of documents on the web showing how to proceed.
1204
+
1205
+ gKnit comes with a simple LaTeX style file for gknitting this blog as a PDF document. Here is
1206
+ the Yaml header to generate this blog in PDF format instead of HTML:
1207
+
1208
+ ```
1209
+ ---
1210
+ title: "gKnit - Ruby and R Knitting with Galaaz"
1211
+ author: "Rodrigo Botafogo"
1212
+ tags: [Galaaz, Ruby, R, JRuby, knitr, gknit]
1213
+ date: "29 October 2018"
1214
+ output:
1215
+ pdf\_document:
1216
+ includes:
1217
+ in\_header: ["../../sty/galaaz.sty"]
1218
+ number\_sections: yes
1219
+ ---
1220
+ ```
1221
+
1222
+ ## Template based documents generation
1223
+
1224
+ When a document is converted to PDF it follows a certain conversion template. We've seen above
1225
+ the use of 'galaaz.sty' as a basic template to generate a PDF document. Using the
1226
+ 'gknit-draft' app that comes with Galaaz, the same .Rmd file can be compiled to different
1227
+ looking PDF documents. Galaaz automatically loads the 'rticles' R package that comes with
1228
+ templates for the following journals with the respective template name:
1229
+
1230
+ * ACM articles: acm_article
1231
+ * ACS articles: acs_article
1232
+ * AEA journal submissions: aea_article
1233
+ * AGU journal submissions: ????
1234
+ * AMS articles: ams_article
1235
+ * American Statistical Association: asa_article
1236
+ * Biometrics articles: biometrics_article
1237
+ * Bulletin de l'AMQ journal submissions: amq_article
1238
+ * CTeX documents: ctex
1239
+ * Elsevier journal submissions: elsevier_article
1240
+ * IEEE Transaction journal submissions: ieee_article
1241
+ * JSS articles: jss_article
1242
+ * MDPI journal submissions: mdpi_article
1243
+ * Monthly Notices of the Royal Astronomical Society articles: mnras_article
1244
+ * NNRAS journal submissions: nmras_article
1245
+ * PeerJ articles: peerj_article
1246
+ * Royal Society Open Science journal submissions: rsos_article
1247
+ * Royal Statistical Society: rss_article
1248
+ * Sage journal submissions: sage_article
1249
+ * Springer journal submissions: springer_article
1250
+ * Statistics in Medicine journal submissions: sim_article
1251
+ * Copernicus Publications journal submissions: copernicus_article
1252
+ * The R Journal articles: rjournal_article
1253
+ * Frontiers articles: ???
1254
+ * Taylor & Francis articles: ???
1255
+ * Bulletin De L'AMQ: amq_article
1256
+ * PLOS journal: plos_article
1257
+ * Proceedings of the National Academy of Sciences of the USA: pnas_article
1258
+
1259
+ In order to create a document with one of those templates, use the following command:
1260
+
1261
+ ```
1262
+ gknit-draft --filename <my_document> --template <template> --package <package>
1263
+ --create_dir
1264
+ ```
1265
+ So, in order to create a template for writing an R Journal, use:
1266
+
1267
+ ```
1268
+ gknit-draft --filename my_r_article --template rjournal_article --package rticles
1269
+ --create_dir
1270
+ ```
1271
+
1272
+ # Accessing R variables
1273
+
1274
+ Galaaz allows Ruby to access variables created in R. For example, the `mtcars` data set is
1275
+ available in R and can be accessed from Ruby by using the tilde operator followed by the
1276
+ symbol for the variable, in this case `:mtcars`. In the code below, method `outputs` is
1277
+ used to output the `mtcars` data set nicely formatted in HTML by use of the `kable` and
1278
+ `kable_styling` functions. Method `outputs` is only available when used with gKnit.
1279
+
1280
+ ```{ruby view_kable}
1281
+ outputs (~R[:mtcars]).kable.kable_styling
1282
+ ```
1283
+
1284
+ # Basic Data Types
1285
+
1286
+ ## Vector
107
1287
 
108
1288
  Vectors can be thought of as contiguous cells containing data. Cells are accessed through
109
1289
  indexing operations such as x[5]. Galaaz has six basic (‘atomic’) vector types: logical,
@@ -116,7 +1296,7 @@ table.
116
1296
  | logical | logical | logical |
117
1297
  | integer | numeric | integer |
118
1298
  | double | numeric | double |
119
- | complex | complex | comples |
1299
+ | complex | complex | complex |
120
1300
  | character | character | character |
121
1301
  | raw | raw | raw |
122
1302
 
@@ -178,7 +1358,7 @@ vec = R.c(true, true, false, false, true)
178
1358
  puts vec
179
1359
  ```
180
1360
 
181
- ## Combining Vectors
1361
+ ### Combining Vectors
182
1362
 
183
1363
  The 'c' functions used to create vectors can also be used to combine two vectors:
184
1364
 
@@ -193,14 +1373,14 @@ In this next example, method 'c' is chainned after 'vec1'. This also looks like
193
1373
  method of the vector, but in reallity, this is actually closer to the pipe operator. When
194
1374
  Galaaz identifies that 'c' is not a method of 'vec' it actually tries to call 'R.c' with
195
1375
  'vec1' as the first argument concatenated with all the other available arguments. The code
196
- bellow is automatically converted to the code above.
1376
+ below is automatically converted to the code above.
197
1377
 
198
1378
  ```{ruby chainning_methods}
199
1379
  vec = vec1.c(vec2)
200
1380
  puts vec
201
1381
  ```
202
1382
 
203
- ## Vector Arithmetic
1383
+ ### Vector Arithmetic
204
1384
 
205
1385
  Arithmetic operations on vectors are performed element by element:
206
1386
 
@@ -219,7 +1399,7 @@ vec3 = R.c(1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0)
219
1399
  puts vec4 = vec1 + vec3
220
1400
  ```
221
1401
 
222
- ## Vector Indexing
1402
+ ### Vector Indexing
223
1403
 
224
1404
  Vectors can be indexed by using the '[]' operator:
225
1405
 
@@ -227,7 +1407,7 @@ Vectors can be indexed by using the '[]' operator:
227
1407
  puts vec4[3]
228
1408
  ```
229
1409
 
230
- We can also index a vector with another vector. For example, in the code bellow, we take elements
1410
+ We can also index a vector with another vector. For example, in the code below, we take elements
231
1411
  1, 3, 5, and 7 from vec3:
232
1412
 
233
1413
  ```{ruby index_by_vector}
@@ -275,7 +1455,7 @@ full_name = R.c(First: "Rodrigo", Middle: "A", Last: "Botafogo")
275
1455
  puts full_name
276
1456
  ```
277
1457
 
278
- ## Extracting Native Ruby Types from a Vector
1458
+ ### Extracting Native Ruby Types from a Vector
279
1459
 
280
1460
  Vectors created with 'R.c' are of class R::Vector. You might have noticed that when indexing a
281
1461
  vector, a new vector is returned, even if this vector has one single element. In order to use
@@ -290,19 +1470,7 @@ puts vec4 >> 4
290
1470
 
291
1471
  Note that indexing with '>>' starts at 0 and not at 1, also, we cannot do negative indexing.
292
1472
 
293
- # Accessing R variables
294
-
295
- Galaaz allows Ruby to access variables created in R. For example, the 'mtcars' data set is
296
- available in R and can be accessed from Ruby by using the 'tilda' operator followed by the
297
- symbol for the variable, in this case ':mtcar'. In the code bellow method 'outputs' is
298
- used to output the 'mtcars' data set nicely formatted in HTML by use of the 'kable' and
299
- 'kable_styling' functions. Method 'outputs' is only available when used with 'gknit'.
300
-
301
- ```{ruby view_kable}
302
- outputs (~:mtcars).kable.kable_styling
303
- ```
304
-
305
- # Matrix
1473
+ ## Matrix
306
1474
 
307
1475
  A matrix is a collection of elements organized as a two dimensional table. A matrix can be
308
1476
  created by the 'matrix' function:
@@ -326,7 +1494,7 @@ mat_row = R.matrix(R.c(1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0),
326
1494
  puts mat_row
327
1495
  ```
328
1496
 
329
- ## Indexing a Matrix
1497
+ ### Indexing a Matrix
330
1498
 
331
1499
  A matrix can be indexed by [row, column]:
332
1500
 
@@ -360,7 +1528,7 @@ and 'cbind':
360
1528
  puts mat_row.cbind(mat)
361
1529
  ```
362
1530
 
363
- # List
1531
+ ## List
364
1532
 
365
1533
  A list is a data structure that can contain sublists of different types, while vector and matrix
366
1534
  can only hold one type of element.
@@ -376,7 +1544,7 @@ puts lst
376
1544
  Note that 'lst' elements are named elements.
377
1545
 
378
1546
 
379
- ## List Indexing
1547
+ ### List Indexing
380
1548
 
381
1549
  List indexing, also called slicing, is done using the '[]' operator and the '[[]]' operator. Let's
382
1550
  first start with the '[]' operator. The list above has three sublist indexing with '[]' will
@@ -406,11 +1574,11 @@ then the first element of the vector was extracted (note that vectors also accep
406
1574
  operator) and then the vector was indexed by its first element, extracting the native Ruby type.
407
1575
 
408
1576
 
409
- # Data Frame
1577
+ ## Data Frame
410
1578
 
411
1579
  A data frame is a table like structure in which each column has the same number of
412
1580
  rows. Data frames are the basic structure for storing data for data analysis. We have already
413
- seen a data frame previously when we accessed variable '~:mtcars'. In order to create a
1581
+ seen a data frame previously when we accessed variable '~R[:mtcars]'. In order to create a
414
1582
  data frame, function 'data__frame' is used:
415
1583
 
416
1584
  ```{ruby dataframe}
@@ -421,51 +1589,51 @@ df = R.data__frame(
421
1589
  puts df
422
1590
  ```
423
1591
 
424
- ## Data Frame Indexing
1592
+ ### Data Frame Indexing
425
1593
 
426
1594
  A data frame can be indexed the same way as a matrix, by using '[row, column]', where row and
427
1595
  column can either be a numeric or the name of the row or column
428
1596
 
429
1597
  ```{ruby dataframe_index}
430
- puts (~:mtcars).head
431
- puts (~:mtcars)[1, 2]
432
- puts (~:mtcars)['Datsun 710', 'mpg']
1598
+ puts (~R[:mtcars]).head
1599
+ puts (~R[:mtcars])[1, 2]
1600
+ puts (~R[:mtcars])['Datsun 710', 'mpg']
433
1601
  ```
434
1602
 
435
1603
  Extracting a column from a data frame as a vector can be done by using the double square bracket
436
1604
  operator:
437
1605
 
438
1606
  ```{ruby dataframe_column}
439
- puts (~:mtcars)[['mpg']]
1607
+ puts (~R[:mtcars])[['mpg']]
440
1608
  ```
441
1609
 
442
1610
  A data frame column can also be accessed as if it were an instance variable of the data frame:
443
1611
 
444
1612
  ```{ruby dataframe_instance_variable}
445
- puts (~:mtcars).mpg
1613
+ puts (~R[:mtcars]).mpg
446
1614
  ```
447
1615
 
448
1616
  Slicing a data frame can be done by indexing it with a vector (we use 'head' to reduce the
449
1617
  output):
450
1618
 
451
1619
  ```{ruby dataframe_column_slice}
452
- puts (~:mtcars)[R.c('mpg', 'hp')].head
1620
+ puts (~R[:mtcars])[R.c('mpg', 'hp')].head
453
1621
  ```
454
1622
 
455
1623
  A row slice can be obtained by indexing by row and using the ':all' keyword for the column:
456
1624
 
457
1625
  ```{ruby dataframe_row_slice}
458
- puts (~:mtcars)[R.c('Datsun 710', 'Camaro Z28'), :all]
1626
+ puts (~R[:mtcars])[R.c('Datsun 710', 'Camaro Z28'), :all]
459
1627
  ```
460
1628
 
461
1629
  Finally, a data frame can also be indexed with a logical vector. In this next example, the
462
1630
  'am' column of :mtcars is compared with 0 (with method 'eq'). When 'am' is equal to 0 the
463
- car is automatic. So, by doing '(~:mtcars).am.eq 0' a logical vector is created with
1631
+ car is automatic. So, by doing '(~R[:mtcars]).am.eq 0' a logical vector is created with
464
1632
  'true' whenever 'am' is 0 and 'false' otherwise.
465
1633
 
466
1634
  ```{ruby logical_vector_filter}
467
1635
  # obtain a vector with 'true' for cars with automatic transmission
468
- automatic = (~:mtcars).am.eq 0
1636
+ automatic = (~R[:mtcars]).am.eq 0
469
1637
  puts automatic
470
1638
  ```
471
1639
 
@@ -474,7 +1642,7 @@ which all cars have automatic transmission.
474
1642
 
475
1643
  ```{ruby dataframe_logical}
476
1644
  # slice the data frame by using this vector
477
- puts (~:mtcars)[automatic, :all]
1645
+ puts (~R[:mtcars])[automatic, :all]
478
1646
  ```
479
1647
 
480
1648
  # Writing Expressions in Galaaz
@@ -484,24 +1652,24 @@ Galaaz extends Ruby to work with complex expressions, similar to R's expressions
484
1652
 
485
1653
  ## Expressions from operators
486
1654
 
487
- The code bellow
1655
+ The code below
488
1656
  creates an expression summing two symbols
489
1657
 
490
1658
  ```{ruby expressions}
491
- exp1 = :a + :b
1659
+ exp1 = R[:a] + R[:b]
492
1660
  puts exp1
493
1661
  ```
494
1662
  We can build any complex mathematical expression
495
1663
 
496
1664
  ```{ruby expr2}
497
- exp2 = (:a + :b) * 2.0 + :c ** 2 / :z
1665
+ exp2 = (R[:a] + R[:b]) * 2.0 + R[:c] ** 2 / R[:z]
498
1666
  puts exp2
499
1667
  ```
500
1668
 
501
1669
  It is also possible to use inequality operators in building expressions
502
1670
 
503
1671
  ```{ruby expr3}
504
- exp3 = (:a + :b) >= :z
1672
+ exp3 = (R[:a] + R[:b]) >= :z
505
1673
  puts exp3
506
1674
  ```
507
1675
 
@@ -510,7 +1678,7 @@ notation for those operators such as (.gt, .ge, etc.). So the same expression w
510
1678
  above can also be written as
511
1679
 
512
1680
  ```{ruby expr4}
513
- exp4 = (:a + :b).ge :z
1681
+ exp4 = (R[:a] + R[:b]).ge :z
514
1682
  puts exp4
515
1683
  ```
516
1684
 
@@ -519,23 +1687,23 @@ those are expressions involving '==', and '='. In order to write an expression
519
1687
  need to use the method '.eq' and for '=' we need the function '.assign'
520
1688
 
521
1689
  ```{ruby expr5}
522
- exp5 = (:a + :b).eq :z
1690
+ exp5 = (R[:a] + R[:b]).eq :z
523
1691
  puts exp5
524
1692
  ```
525
1693
 
526
1694
  ```{ruby expr6}
527
- exp6 = :y.assign :a + :b
1695
+ exp6 = R[:y].assign R[:a] + R[:b]
528
1696
  puts exp6
529
1697
  ```
530
1698
  In general we think that using the functional notation is preferable to using the
531
1699
  symbolic notation as otherwise, we end up writing invalid expressions such as
532
1700
 
533
- ```{ruby exp_wrong, warning=FALSE}
534
- exp_wrong = (:a + :b) == :z
1701
+ ```{ruby exp_wrong, warning=FALSE, eval=FALSE}
1702
+ exp_wrong = (R[:a] + R[:b]) == :z
535
1703
  puts exp_wrong
536
1704
  ```
537
1705
  and it might be difficult to understand what is going on here. The problem lies with the fact that
538
- when using '==' we are comparing expression (:a + :b) to expression :z with '=='. When the
1706
+ when using '==' we are comparing expression (R[:a] + R[:b]) to expression :z with '=='. When the
539
1707
  comparison is executed, the system tries to evaluate :a, :b and :z, and those symbols at
540
1708
  this time are not bound to anything and we get a "object 'a' not found" message.
541
1709
  If we only use functional notation, this type of error will not occur.
@@ -550,21 +1718,21 @@ When we want the function to be part of the expression, we call the function pre
550
1718
  by the letter E, such as 'E.sin(x)'
551
1719
 
552
1720
  ```{ruby method_expression}
553
- exp7 = :y.assign E.sin(:x)
1721
+ exp7 = R[:y].assign E.sin(R[:x])
554
1722
  puts exp7
555
1723
  ```
556
1724
 
557
1725
  Expressions can also be written using '.' notation:
558
1726
 
559
1727
  ```{ruby expression_with_dot}
560
- exp8 = :y.assign :x.sin
1728
+ exp8 = R[:y].assign R[:x].sin
561
1729
  puts exp8
562
1730
  ```
563
1731
 
564
1732
  When a function has multiple arguments, the first one can be used before the '.':
565
1733
 
566
1734
  ```{ruby expression_multiple_args}
567
- exp9 = :x.c(:y)
1735
+ exp9 = R[:x].c(R[:y])
568
1736
  puts exp9
569
1737
  ```
570
1738
 
@@ -574,7 +1742,7 @@ Expressions can be evaluated by calling function 'eval' with a binding. A bindin
574
1742
  with a list:
575
1743
 
576
1744
  ```{ruby eval_expression_list}
577
- exp = (:a + :b) * 2.0 + :c ** 2 / :z
1745
+ exp = (R[:a] + R[:b]) * 2.0 + R[:c] ** 2 / R[:z]
578
1746
  puts exp.eval(R.list(a: 10, b: 20, c: 30, z: 40))
579
1747
  ```
580
1748
 
@@ -593,35 +1761,39 @@ puts exp.eval(df)
593
1761
  # Manipulating Data
594
1762
 
595
1763
  One of the major benefits of Galaaz is to bring strong data manipulation to Ruby. The following
596
- examples were extracted from Hardley's "R for Data Science" (https://r4ds.had.co.nz/). This
1764
+ examples were extracted from Hadley's "R for Data Science" (https://r4ds.had.co.nz/). This
597
1765
  is a highly recommended book for those not already familiar with the 'tidyverse' style of
598
1766
  programming in R. In the sections to follow, we will limit ourselves to convert the R code to
599
1767
  Galaaz.
600
1768
 
601
1769
  For these
602
1770
  examples, we will investigate the nycflights13 data set available on the package by the
603
- same name. We use function 'R.install_and_loads' that checks if the library is available
1771
+ same name. We use function 'R.install\_and\_loads' that checks if the library is available
604
1772
  locally, and if not, installs it. This data frame contains all 336,776 flights that
605
1773
  departed from New York City in 2013. The data comes from the US Bureau of
606
1774
  Transportation Statistics.
607
1775
 
1776
+ Dplyr often uses **tibbles** in place of classic data frames. In Galaaz, printing may differ from
1777
+ the R console; if you need a classic tabular printout, convert with **`as__data__frame`** (or use
1778
+ `head` / `str` in R via `R` calls).
1779
+
608
1780
  ```{ruby nycflights13}
609
1781
  R.install_and_loads('nycflights13')
610
1782
  R.library('dplyr')
611
1783
  ```
612
1784
 
613
1785
  ```{ruby flights}
614
- flights = ~:flights
615
- puts flights.head.as__data__frame
1786
+ flights = ~R[:flights]
1787
+ puts flights.head
616
1788
  ```
617
1789
 
618
1790
  ## Filtering rows with Filter
619
1791
 
620
1792
  In this example we filter the flights data set by giving to the filter function two expressions:
621
- the first :month.eq 1
1793
+ the first R[:month].eq 1
622
1794
 
623
1795
  ```{ruby filter_rows}
624
- puts flights.filter((:month.eq 1), (:day.eq 1)).head.as__data__frame
1796
+ puts flights.filter((R[:month].eq 1), (R[:day].eq 1)).head
625
1797
  ```
626
1798
 
627
1799
  ## Logical Operators
@@ -629,7 +1801,7 @@ puts flights.filter((:month.eq 1), (:day.eq 1)).head.as__data__frame
629
1801
  All flights that departed in November of December
630
1802
 
631
1803
  ```{ruby nov_dec}
632
- puts flights.filter((:month.eq 11) | (:month.eq 12)).head.as__data__frame
1804
+ puts flights.filter((R[:month].eq 11) | (R[:month].eq 12)).head
633
1805
  ```
634
1806
 
635
1807
  The same as above, but using the 'in' operator. In R, it is possible to define many operators
@@ -638,7 +1810,7 @@ operators from Galaaz the '._' method is used, where the first argument is the o
638
1810
  symbol, in this case ':in' and the second argument is the vector:
639
1811
 
640
1812
  ```{ruby in_op}
641
- puts flights.filter(:month._ :in, R.c(11, 12)).head.as__data__frame
1813
+ puts flights.filter(R[:month]._ :in, R.c(11, 12)).head
642
1814
  ```
643
1815
 
644
1816
  ## Filtering with NA (Not Available)
@@ -650,20 +1822,20 @@ what is obtained from data frame.
650
1822
 
651
1823
  ```{ruby na_tibble}
652
1824
  df = R.tibble(x: R.c(1, R::NA, 3))
653
- puts df.as__data__frame
1825
+ puts df
654
1826
  ```
655
1827
 
656
1828
  Now filtering by :x > 1 shows all lines that satisfy this condition, where the row with R:NA does
657
1829
  not.
658
1830
 
659
1831
  ```{ruby filter_na}
660
- puts df.filter(:x > 1).as__data__frame
1832
+ puts df.filter(R[:x] > 1)
661
1833
  ```
662
1834
 
663
1835
  To match an NA use method 'is__na'
664
1836
 
665
1837
  ```{ruby with_na}
666
- puts df.filter((:x.is__na) | (:x > 1)).as__data__frame
1838
+ puts df.filter((R[:x].is__na) | (R[:x] > 1))
667
1839
  ```
668
1840
 
669
1841
  ## Arrange Rows with arrange
@@ -671,13 +1843,13 @@ puts df.filter((:x.is__na) | (:x > 1)).as__data__frame
671
1843
  Arrange reorders the rows of a data frame by the given arguments.
672
1844
 
673
1845
  ```{ruby arrange}
674
- puts flights.arrange(:year, :month, :day).head.as__data__frame
1846
+ puts flights.arrange(:year, :month, :day).head
675
1847
  ```
676
1848
 
677
1849
  To arrange in descending order, use function 'desc'
678
1850
 
679
1851
  ```{ruby desc_arrange}
680
- puts flights.arrange(:dep_delay.desc).head.as__data__frame
1852
+ puts flights.arrange(R[:dep_delay].desc).head
681
1853
  ```
682
1854
 
683
1855
  ## Selecting columns
@@ -685,19 +1857,19 @@ puts flights.arrange(:dep_delay.desc).head.as__data__frame
685
1857
  To select specific columns from a dataset we use function 'select':
686
1858
 
687
1859
  ```{ruby select}
688
- puts flights.select(:year, :month, :day).head.as__data__frame
1860
+ puts flights.select(:year, :month, :day).head
689
1861
  ```
690
1862
 
691
1863
  It is also possible to select column in a given range
692
1864
 
693
1865
  ```{ruby select_range}
694
- puts flights.select(:year.up_to :day).head.as__data__frame
1866
+ puts flights.select(R[:year].up_to(R[:day])).head
695
1867
  ```
696
1868
 
697
1869
  Select all columns that start with a given name sequence
698
1870
 
699
1871
  ```{ruby select_starts_with}
700
- puts flights.select(E.starts_with('arr')).head.as__data__frame
1872
+ puts flights.select(E.starts_with('arr')).head
701
1873
  ```
702
1874
 
703
1875
  Other functions that can be used:
@@ -714,26 +1886,26 @@ Other functions that can be used:
714
1886
  A helper function that comes in handy when we just want to rearrange column order is 'Everything':
715
1887
 
716
1888
  ```{ruby everything}
717
- puts flights.select(:year, :month, :day, E.everything).head.as__data__frame
1889
+ puts flights.select(:year, :month, :day, E.everything).head
718
1890
  ```
719
1891
 
720
1892
  ## Add variables to a dataframe with 'mutate'
721
1893
 
722
1894
  ```{ruby small_flights}
723
1895
  flights_sm = flights.
724
- select((:year.up_to :day),
1896
+ select((R[:year].up_to(R[:day])),
725
1897
  E.ends_with('delay'),
726
1898
  :distance,
727
1899
  :air_time)
728
1900
 
729
- puts flights_sm.head.as__data__frame
1901
+ puts flights_sm.head
730
1902
  ```
731
1903
 
732
1904
  ```{ruby mutate}
733
1905
  flights_sm = flights_sm.
734
- mutate(gain: :dep_delay - :arr_delay,
735
- speed: :distance / :air_time * 60)
736
- puts flights_sm.head.as__data__frame
1906
+ mutate(gain: R[:dep_delay] - R[:arr_delay],
1907
+ speed: R[:distance] / R[:air_time] * 60)
1908
+ puts flights_sm.head
737
1909
  ```
738
1910
 
739
1911
  ## Summarising data
@@ -742,14 +1914,14 @@ Function 'summarise' calculates summaries for the data frame. When no 'group_by'
742
1914
  a single value is obtained from the data frame:
743
1915
 
744
1916
  ```{ruby summarise}
745
- puts flights.summarise(delay: E.mean(:dep_delay, na__rm: true)).as__data__frame
1917
+ puts flights.summarise(delay: E.mean(:dep_delay, na__rm: true))
746
1918
  ```
747
1919
 
748
- When a data frame is groupe with 'group_by' summaries apply to the given group:
1920
+ When a data frame is grouped with 'group_by' summaries apply to the given group:
749
1921
 
750
1922
  ```{ruby summarise_group_by}
751
1923
  by_day = flights.group_by(:year, :month, :day)
752
- puts by_day.summarise(delay: :dep_delay.mean(na__rm: true)).head.as__data__frame
1924
+ puts by_day.summarise(delay: R[:dep_delay].mean(na__rm: true)).head
753
1925
  ```
754
1926
 
755
1927
  Next we put many operations together by pipping them one after the other:
@@ -759,23 +1931,25 @@ delays = flights.
759
1931
  group_by(:dest).
760
1932
  summarise(
761
1933
  count: E.n,
762
- dist: :distance.mean(na__rm: true),
763
- delay: :arr_delay.mean(na__rm: true)).
764
- filter(:count > 20, :dest != "NHL")
1934
+ dist: R[:distance].mean(na__rm: true),
1935
+ delay: R[:arr_delay].mean(na__rm: true)).
1936
+ filter(R[:count] > 20, R[:dest] != "NHL")
765
1937
 
766
- puts delays.as__data__frame.head
1938
+ puts delays.head
767
1939
  ```
768
1940
 
769
1941
  # Using Data Table
770
1942
 
1943
+ The next chunk converts the **nycflights13** `flights` tibble already loaded above into a
1944
+ **`data.table`**. That keeps the manual offline and avoids downloading a remote CSV during gknit
1945
+ (network stalls look like bridge hangs when the transfer runs inside a single R eval).
1946
+
771
1947
  ```{ruby fread}
772
1948
  R.library('data.table')
773
- R.install_and_loads('curl')
774
1949
 
775
- input = "https://raw.githubusercontent.com/Rdatatable/data.table/master/vignettes/flights14.csv"
776
- flights = R.fread(input)
777
- puts flights
1950
+ flights = R.as__data__table(~R[:flights])
778
1951
  puts flights.dim
1952
+ puts R.head(flights, 12)
779
1953
  ```
780
1954
 
781
1955
  ```{ruby data_table}
@@ -793,7 +1967,7 @@ puts data_table.ID
793
1967
 
794
1968
  ```{ruby subset_i}
795
1969
  # subset rows in i
796
- ans = flights[(:origin.eq "JFK") & (:month.eq 6)]
1970
+ ans = flights[(R[:origin].eq "JFK") & (R[:month].eq 6)]
797
1971
  puts ans.head
798
1972
 
799
1973
  # Get the first two rows from flights.
@@ -801,8 +1975,7 @@ puts ans.head
801
1975
  ans = flights[(1..2)]
802
1976
  puts ans
803
1977
 
804
- # Sort flights first by column origin in ascending order, and then by dest in descending order:
805
-
1978
+ # Sort by origin asc, then dest desc (example kept commented):
806
1979
  # ans = flights[E.order(:origin, -(:dest))]
807
1980
  # puts ans.head
808
1981
 
@@ -815,66 +1988,274 @@ puts ans
815
1988
  ans = flights[:all, :arr_delay]
816
1989
  puts ans.head
817
1990
 
818
- # Select arr_delay column, but return as a data.table instead.
1991
+ # arr_delay as data.table (not plain vector).
819
1992
 
820
- ans = flights[:all, :arr_delay.list]
1993
+ ans = flights[:all, R[:arr_delay].list]
821
1994
  puts ans.head
822
1995
 
823
- ans = flights[:all, E.list(:arr_delay, :dep_delay)]
1996
+ ans = flights[:all, E.list(R[:arr_delay], R[:dep_delay])]
1997
+ ```
1998
+
1999
+ # Apache Arrow
2000
+
2001
+ [Apache Arrow](https://arrow.apache.org/) is a **columnar** in-memory format used heavily in R
2002
+ and Python for analytics. In Galaaz, **Ruby does not hold an Arrow C++ table itself**; instead you
2003
+ build ordinary Ruby structures (arrays of row hashes), and **`R::Arrow.from_ruby_batches`** creates
2004
+ a real **Arrow `Table` inside GNU R**. From there you use R’s **`arrow`** and **`dplyr`** packages
2005
+ as usual: **`group_by`** on the Arrow table, **`summarise`** for aggregates, then **`collect()`** to
2006
+ materialize a tibble when you need in-memory R rows.
2007
+
2008
+ That pattern matches production use: **JRuby threads** (or sequential code) assemble many rows in
2009
+ Ruby; you pay **one** bridge-heavy handoff to R; **dplyr** runs vectorised work on the Arrow table
2010
+ in R.
2011
+
2012
+ **Prerequisites:** install R packages **`arrow`** and **`dplyr`**. Run scripts with
2013
+ **`bin/galaaz-jruby`** (or the same JVM flags as in **`docs/testing.md`**) so the Arrow JNI stack is
2014
+ available.
2015
+
2016
+ ## Other `R::Arrow` helpers
2017
+
2018
+ The Ruby module **`R::Arrow`** (see `lib/R_interface/r_arrow.rb`) also includes:
2019
+
2020
+ * **`R::Arrow.table_from(df)`** — wrap an R `data.frame` / tibble as an Arrow table.
2021
+ * **`R::Arrow.read_feather` / `write_feather`**, **`read_parquet`**, **`dataset(path)`** — file and
2022
+ dataset IO on paths visible to R.
2023
+
2024
+ ## Example: many Ruby rows → Arrow in R → grouped statistics
2025
+
2026
+ The repository test **`slow-specs/arrow_large_pipeline_spec.rb`** builds **200k rows** in parallel
2027
+ (eight threads × 25,000 rows), pushes them through **`R::Arrow.from_ruby_batches`**, then checks that
2028
+ **dplyr** group summaries match a Ruby reference calculation. The same logic appears below at a
2029
+ **smaller scale** so this manual can knit quickly; increase `thread_count` and `rows_per_thread`
2030
+ when experimenting locally.
2031
+
2032
+ ```{ruby arrow_pipeline_example, message=FALSE, warning=FALSE}
2033
+ # Scaled-down version of slow-specs/arrow_large_pipeline_spec.rb.
2034
+ unless R::Support.eval("requireNamespace('arrow', quietly=TRUE) && requireNamespace('dplyr', quietly=TRUE)") == true
2035
+ puts '(Skip: need arrow + dplyr in R; use bin/galaaz-jruby outside gKnit.)'
2036
+ else
2037
+ thread_count = 4
2038
+ rows_per_thread = 500
2039
+ group_count = 5
2040
+
2041
+ batches = []
2042
+ mutex = Mutex.new
2043
+ threads = []
2044
+
2045
+ thread_count.times do |tid|
2046
+ threads << Thread.new do
2047
+ start = tid * rows_per_thread
2048
+ local = (start...(start + rows_per_thread)).map do |i|
2049
+ {
2050
+ id: i,
2051
+ grp: "g#{i % group_count}",
2052
+ value: (i % 17) + 1,
2053
+ weight: ((i % 5) + 1) * 0.5
2054
+ }
2055
+ end
2056
+ mutex.synchronize { batches << local }
2057
+ end
2058
+ end
2059
+ threads.each(&:join)
2060
+
2061
+ tbl = R::Arrow.from_ruby_batches(batches)
2062
+ puts "R class after from_ruby_batches: #{tbl.rclass}"
2063
+
2064
+ grouped = R.dplyr___group_by(tbl, :grp)
2065
+ summarised = R.dplyr___summarise(
2066
+ grouped,
2067
+ n: E.n(),
2068
+ total: E.sum(:value),
2069
+ wsum: E.sum(R[:value] * R[:weight])
2070
+ )
2071
+ out = R.dplyr___collect(summarised)
2072
+
2073
+ puts 'Per-group summary (first rows):'
2074
+ puts R.as__data__frame(out).head(10)
2075
+
2076
+ total_n = 0
2077
+ (1..(out.nrow >> 0)).each { |i| total_n += (out[['n']][i] >> 0) }
2078
+ puts "Sum of group counts n (should equal #{thread_count * rows_per_thread}): #{total_n}"
2079
+ end
2080
+ ```
2081
+
2082
+ **What to notice:** (1) Ruby only sees **`Hash`** rows and Ruby **`Thread`** objects; (2) a single
2083
+ **`from_ruby_batches`** call creates the Arrow table in R; (3) **`dplyr___group_by`** /
2084
+ **`dplyr___summarise`** / **`dplyr___collect`** mirror **`dplyr::group_by`** /
2085
+ **`dplyr::summarise`** / **`dplyr::collect`** on an Arrow-backed table. For a lighter test, see
2086
+ **`specs/arrow_from_ruby_batches_spec.rb`**; for the full-size benchmark, run
2087
+ **`bin/run_slow_rspec slow-specs/arrow_large_pipeline_spec.rb`**.
2088
+
2089
+ # Bioconductor and DESeq2
2090
+
2091
+ **Bioconductor** packages are ordinary R packages installed from the Bioconductor repositories.
2092
+ Galaaz does not treat them specially: once installed in **GNU R**, you load them with
2093
+ **`R.library`** like any CRAN package.
2094
+
2095
+ ## Installing Bioconductor packages
2096
+
2097
+ From an R session (or `R -e '...'`), use **BiocManager** (see
2098
+ [bioconductor.org](https://bioconductor.org/install/)):
2099
+
2100
+ ```r
2101
+ if (!requireNamespace("BiocManager", quietly = TRUE))
2102
+ install.packages("BiocManager")
2103
+ BiocManager::install(c("DESeq2", "airway"))
2104
+ ```
2105
+
2106
+ The **`airway`** package ships the example **`SummarizedExperiment`** used below. **DESeq2**
2107
+ pulls in several dependencies; the first install can take several minutes.
2108
+
2109
+ ## Example: DESeq2 on the airway dataset
2110
+
2111
+ The script **`examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb`** is the canonical
2112
+ version in the repository. Run it from the **Galaaz repository root** with JRuby, for example:
2113
+
2114
+ ```text
2115
+ bin/galaaz-jruby examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb
2116
+ ```
2117
+
2118
+ The workflow in Ruby mirrors a standard DESeq2 vignette:
2119
+
2120
+ 1. **`R.library('DESeq2')`** and **`R.library('airway')`**, then **`R.data('airway')`** so the
2121
+ object exists in R’s global environment.
2122
+ 2. **`airway = ~R[:airway]`** pulls the experiment into a Galaaz wrapper so you can pass it to R
2123
+ functions as a Ruby value.
2124
+ 3. **`R.DESeqDataSet(..., design: (R[:all].til R[:cell] + R[:dex]))`** builds the **`DESeqDataSet`**. The
2125
+ **`(R[:all].til R[:cell] + R[:dex])`** form is Galaaz’s way of passing the one-sided formula
2126
+ **`~ cell + dex`** (adjust for the design you need).
2127
+ 4. Prefilter rows with almost no counts: **`keep = R.rowSums(R.counts(dds)) >= 10`** and
2128
+ **`dds = dds[keep, :all]`**.
2129
+ 5. **`dds = R.DESeq(dds)`** fits the model; **`res = R.results(dds, contrast: R.c('dex', 'trt', 'untrt'))`**
2130
+ extracts the treatment contrast (adjust **`contrast`** for your experiment).
2131
+ 6. Summaries use normal Ruby string interpolation on **`R.nrow`**, **`R.ncol`**, **`R.colnames`**, etc.
2132
+ 7. **`R.pdf(...); R.plotMA(res, ...); R.dev__off`** writes DESeq2’s MA plot (path is relative to the
2133
+ process working directory—use the repo root when running the bundled script).
2134
+
2135
+ Related benchmarks and warm-run notes live under **`docs/deseq2_airway_benchmark.md`** and
2136
+ **`examples/bioconductor_deseq2_airway/bench_*.rb`**.
2137
+
2138
+ Below is the full listing (same as the file in the repository). It is **not** executed while this
2139
+ manual is knitted, because **DESeq2** is heavy and may be absent on the build machine.
2140
+
2141
+ ```{ruby deseq2_airway_full_listing, eval=FALSE}
2142
+ # Canonical script: examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb
2143
+ # Run: bin/galaaz-jruby examples/.../deseq2_airway_galaaz.rb (repo root).
2144
+
2145
+ require 'galaaz'
2146
+
2147
+ R.library('DESeq2')
2148
+ R.library('airway')
2149
+ R.data('airway')
2150
+
2151
+ airway = ~R[:airway]
2152
+
2153
+ # Build DESeq2 dataset with one-sided formula: ~ cell + dex.
2154
+ dds = R.DESeqDataSet(airway, design: (R[:all].til R[:cell] + R[:dex]))
2155
+
2156
+ # Prefilter genes with almost no counts.
2157
+ keep = R.rowSums(R.counts(dds)) >= 10
2158
+ dds = dds[keep, :all]
2159
+
2160
+ # Fit DE model and extract treatment effect.
2161
+ dds = R.DESeq(dds)
2162
+ res = R.results(dds, contrast: R.c('dex', 'trt', 'untrt'))
2163
+
2164
+ # Compact sanity outputs for quick verification.
2165
+ puts "Samples: #{R.ncol(dds)}"
2166
+ puts "Genes after prefilter: #{R.nrow(dds)}"
2167
+ puts "Result rows: #{R.nrow(res)}"
2168
+ puts "Result columns: #{R.colnames(res)}"
2169
+ puts "Significant genes (padj < 0.05): #{R.sum(res.padj < 0.05, na__rm: true)}"
2170
+
2171
+ res_ordered = res[R.order(res.padj), :all]
2172
+ puts R.head(R.as__data__frame(res_ordered), 10)
2173
+
2174
+ # Standard DESeq2 plot call written to file.
2175
+ R.pdf('examples/bioconductor_deseq2_airway/plotMA_galaaz.pdf')
2176
+ R.plotMA(res, ylim: R.c(-5, 5))
2177
+ R.dev__off
2178
+ ```
2179
+
2180
+ If **DESeq2** and **airway** are installed, the next chunk loads the data and prints a short
2181
+ preview (it does **not** run **`DESeq`** so the manual knits quickly).
2182
+
2183
+ ```{ruby deseq2_airway_smoke, message=FALSE, warning=FALSE}
2184
+ unless R::Support.eval("requireNamespace('DESeq2', quietly=TRUE) && requireNamespace('airway', quietly=TRUE)")
2185
+ puts '(Skip: install DESeq2 and airway via BiocManager in R to run the full example.)'
2186
+ else
2187
+ R.library('DESeq2')
2188
+ R.library('airway')
2189
+ R.data('airway')
2190
+ airway = ~R[:airway]
2191
+ puts 'airway object (head of assay / dims via R):'
2192
+ puts "ncol(samples): #{R.ncol(airway)}"
2193
+ puts R.head(R.assay(airway), 3)
2194
+ end
824
2195
  ```
825
2196
 
2197
+ # Performance
2198
+
2199
+ For realistic analyses, **most wall-clock time is spent inside GNU R** (model fitting, I/O inside
2200
+ R, graphics). The Galaaz **bridge** adds overhead mainly from **starting a session**, **serializing
2201
+ requests**, and **wrapping results** in Ruby objects—not from reimplementing R’s numerical work.
2202
+
2203
+ Practical tips:
2204
+
2205
+ * Keep **hot loops** in R or vectorized code when possible; use Ruby for orchestration, I/O, and
2206
+ glue.
2207
+ * **Reuse one process**: running many short scripts cold-starts Ruby, the JVM, and R each time;
2208
+ a long-lived process or repeated calls in one run amortize setup (see benchmarks below).
2209
+ * **Batch data**: merge shards in Ruby, then call **`R::Arrow.from_ruby_batches`** (or build one
2210
+ data frame) instead of millions of tiny R calls.
2211
+
2212
+ For measured discussion (including DESeq2-style workloads and warm comparisons), see
2213
+ **`docs/performance.md`** and **`docs/deseq2_airway_benchmark.md`** in the Galaaz repository.
2214
+
826
2215
  # Graphics in Galaaz
827
2216
 
828
2217
  Creating graphics in Galaaz is quite easy, as it can use all the power of ggplot2. There are
829
- many resources in the web that teaches ggplot, so here we give a quick example of ggplot
2218
+ many resources on the web that teach ggplot, so here we give a quick example of ggplot
830
2219
  integration with Ruby. We continue to use the :mtcars dataset and we will plot a diverging
831
- bar plot, showing cars that have 'above' or 'below' gas consuption. Let's first prepare
2220
+ bar plot, showing cars that have 'above' or 'below' gas consumption. Let's first prepare
832
2221
  the data frame with the necessary data:
833
2222
 
834
2223
  ```{ruby diverging_plot_pre}
835
- # copy the R variable :mtcars to the Ruby mtcars variable
836
- mtcars = ~:mtcars
837
-
838
- # create a new column 'car_name' to store the car names so that it can be
839
- # used for plotting. The 'rownames' of the data frame cannot be used as
840
- # data for plotting
841
- mtcars.car_name = R.rownames(:mtcars)
842
-
843
- # compute normalized mpg and add it to a new column called mpg_z
844
- # Note that the mean value for mpg can be obtained by calling the 'mean'
845
- # function on the vector 'mtcars.mpg'. The same with the standard
846
- # deviation 'sd'. The vector is then rounded to two digits with 'round 2'
2224
+ # :mtcars -> Ruby handle
2225
+ mtcars = ~R[:mtcars]
2226
+
2227
+ # Row labels are not a plot column; copy them to car_name.
2228
+ mtcars.car_name = R.rownames(R[:mtcars])
2229
+
2230
+ # Z-score mpg (mean/sd on mtcars.mpg); round to 2 decimals.
847
2231
  mtcars.mpg_z = ((mtcars.mpg - mtcars.mpg.mean)/mtcars.mpg.sd).round 2
848
2232
 
849
- # create a new column 'mpg_type'. Function 'ifelse' is a vectorized function
850
- # that looks at every element of the mpg_z vector and if the value is below
851
- # 0, returns 'below', otherwise returns 'above'
2233
+ # ifelse is vectorized: below / above average mpg_z.
852
2234
  mtcars.mpg_type = (mtcars.mpg_z < 0).ifelse("below", "above")
853
2235
 
854
- # order the mtcar data set by the mpg_z vector from smaler to larger values
2236
+ # Sort rows by mpg_z.
855
2237
  mtcars = mtcars[mtcars.mpg_z.order, :all]
856
2238
 
857
- # convert the car_name column to a factor to retain sorted order in plot
2239
+ # Factor car_name so plot order follows sort.
858
2240
  mtcars.car_name = mtcars.car_name.factor levels: mtcars.car_name
859
2241
 
860
- # let's look at the final data frame
861
2242
  puts mtcars.head
862
2243
  ```
863
- Now, lets plot the diverging bar plot. When using gKnit, there is no need to call
864
- 'R.awt' to create a plotting device, since gKnit does take care of it. Galaaz
2244
+ Now, let's plot the diverging bar plot. When using gKnit, you normally do **not** need to open a
2245
+ graphics device manually; gKnit arranges the figure device for chunk output. Galaaz
865
2246
  provides integration with ggplot. The interested reader should check online for more
866
2247
  information on ggplot, since it is outside the scope of this manual describing
867
- how ggplot works. We give here but a brief description on how this plot is generated.
2248
+ how ggplot works. Here we give only a brief description of how this plot is generated.
868
2249
 
869
- ggplot implements the 'grammar of graphics'. In this approach, plots are build by
2250
+ ggplot implements the 'grammar of graphics'. In this approach, plots are built by
870
2251
  adding layers to the plot. On the first layer we describe what we want on the 'x'
871
2252
  and 'y' axis of the plot. In this case, we have 'car_name' on the 'x' axis and
872
2253
  'mpg\_z' on the 'y' axis. Then the type of graph is specified by adding
873
2254
  'geom\_bar' (for a bar graph). We specify that our bars should be filled using
874
- 'mpg\_type', which is either 'above' or 'bellow' giving then two colours for
2255
+ 'mpg\_type', which is either 'above' or 'below' giving then two colours for
875
2256
  filling. On the next layer we specify the labels for the graph, then we add the
876
2257
  title and subtitle. Finally, in a bar chart usually bars go on the vertical direction,
877
- but in this graph we want the bars to be horizontally layed so we add 'coord\_flip'.
2258
+ but in this graph we want the bars to be horizontally laid so we add 'coord\_flip'.
878
2259
 
879
2260
  ```{ruby diverging_bar, fig.width = 9.1, fig.height = 6.5}
880
2261
  require 'ggplot'
@@ -892,7 +2273,7 @@ puts mtcars.ggplot(E.aes(x: :car_name, y: :mpg_z, label: :mpg_z)) +
892
2273
  # Coding with Tidyverse
893
2274
 
894
2275
  In R, and when coding with 'tidyverse', arguments to a function are usually not
895
- *referencially transparent*. That is, you can’t replace a value with a seemingly equivalent
2276
+ *referentially transparent*. That is, you can’t replace a value with a seemingly equivalent
896
2277
  object that you’ve defined elsewhere. To see the problem, let's first define a data frame:
897
2278
 
898
2279
  ```{ruby df}
@@ -908,17 +2289,17 @@ filter(df, my_var == 1)
908
2289
  ```
909
2290
  It generates the following error: "object 'x' not found.
910
2291
 
911
- However, in Galaaz, arguments are referencially transparent as can be seen by the
912
- code bellow. Note initally that 'my_var = :x' will not give the error "object 'x' not found"
2292
+ However, in Galaaz, arguments are referentially transparent as can be seen by the
2293
+ code below. Note initially that 'my_var = R[:x]' will not give the error "object 'x' not found"
913
2294
  since ':x' is treated as an expression and assigned to my\_var. Then when doing (my\_var.eq 1),
914
- my\_var is a variable that resolves to ':x' and it becomes equivalent to (:x.eq 1) which is
2295
+ my\_var is a variable that resolves to ':x' and it becomes equivalent to (R[:x].eq 1) which is
915
2296
  what we want.
916
2297
 
917
2298
  ```{ruby my_var}
918
- my_var = :x
2299
+ my_var = R[:x]
919
2300
  puts df.filter(my_var.eq 1)
920
2301
  ```
921
- As stated by Hardley
2302
+ As stated by Hadley
922
2303
 
923
2304
  > dplyr code is ambiguous. Depending on what variables are defined where,
924
2305
  > filter(df, x == y) could be equivalent to any of:
@@ -930,9 +2311,9 @@ df[x == df$y, ]
930
2311
  df[x == y, ]
931
2312
  ```
932
2313
  In galaaz this ambiguity does not exist, filter(df, x.eq y) is not a valid expression as
933
- expressions are build with symbols. In doing filter(df, :x.eq y) we are looking for elements
2314
+ expressions are build with symbols. In doing filter(df, R[:x].eq y) we are looking for elements
934
2315
  of the 'x' column that are equal to a previously defined y variable. Finally in
935
- filter(df, :x.eq :y) we are looking for elements in which the 'x' column value is equal to
2316
+ filter(df, R[:x].eq R[:y]) we are looking for elements in which the 'x' column value is equal to
936
2317
  the 'y' column value. This can be seen in the following two chunks of code:
937
2318
 
938
2319
  ```{ruby disamb1}
@@ -940,13 +2321,13 @@ y = 1
940
2321
  x = 2
941
2322
 
942
2323
  # looking for values where the 'x' column is equal to the 'y' column
943
- puts df.filter(:x.eq :y)
2324
+ puts df.filter(R[:x].eq R[:y])
944
2325
  ```
945
2326
 
946
2327
  ```{ruby disamb2}
947
2328
  # looking for values where the 'x' column is equal to the 'y' variable
948
2329
  # in this case, the number 1
949
- puts df.filter(:x.eq y)
2330
+ puts df.filter(R[:x].eq y)
950
2331
  ```
951
2332
  ## Writing a function that applies to different data sets
952
2333
 
@@ -973,11 +2354,12 @@ Unfortunately, in R, this function can fail silently if one of the variables isn
973
2354
  in the data frame, but is present in the global environment. We will not go through here how
974
2355
  to solve this problem in R.
975
2356
 
976
- In Galaaz the method mutate_y bellow will work fine and will never fail silently.
2357
+ In Galaaz the method mutate_y below will work fine and will never fail silently.
977
2358
 
978
2359
  ```{ruby mutate_y, warning=FALSE}
979
2360
  def mutate_y(df)
980
- df.mutate(:y.assign :a + :x)
2361
+ # Mutate column names are Ruby kwargs (y: …). Use .assign only for R `<-` expressions.
2362
+ df.mutate(y: R[:a] + R[:x])
981
2363
  end
982
2364
  ```
983
2365
  Here we create a data frame that has only one column named 'x':
@@ -987,8 +2369,8 @@ df1 = R.data__frame(x: (1..3))
987
2369
  puts df1
988
2370
  ```
989
2371
 
990
- Note that method mutate_y will fail independetly from the fact that variable 'a' is defined and
991
- in the scope of the method. Variable 'a' has no relationship with the symbol ':a' used in the
2372
+ Note that method mutate_y will fail independently from the fact that variable 'a' is defined and
2373
+ in the scope of the method. Variable 'a' has no relationship with the symbol `R[:a]` used in the
992
2374
  definition of 'mutate\_y' above:
993
2375
 
994
2376
  ```{ruby call_mutate_y, warning = FALSE}
@@ -997,12 +2379,13 @@ mutate_y(df1)
997
2379
  ```
998
2380
  ## Different expressions
999
2381
 
1000
- Let's move to the next problem as presented by Hardley where trying to write a function in R
2382
+ Let's move to the next problem as presented by Hadley where trying to write a function in R
1001
2383
  that will receive two argumens, the first a variable and the second an expression is not trivial.
1002
- Bellow we create a data frame and we want to write a function that groups data by a variable and
2384
+ Below we create a data frame and we want to write a function that groups data by a variable and
1003
2385
  summarises it by an expression:
1004
2386
 
1005
2387
  ```{r diff_expr}
2388
+ library(dplyr)
1006
2389
  set.seed(123)
1007
2390
 
1008
2391
  df <- data.frame(
@@ -1027,7 +2410,7 @@ d2 <- df %>%
1027
2410
  as.data.frame(d2)
1028
2411
  ```
1029
2412
 
1030
- As shown by Hardley, one might expect this function to do the trick:
2413
+ As shown by Hadley, one might expect this function to do the trick:
1031
2414
 
1032
2415
  ```{r diff_exp_fnc}
1033
2416
  my_summarise <- function(df, group_var) {
@@ -1042,32 +2425,32 @@ my_summarise <- function(df, group_var) {
1042
2425
 
1043
2426
  In order to solve this problem, coding with dplyr requires the introduction of many new concepts
1044
2427
  and functions such as 'quo', 'quos', 'enquo', 'enquos', '!!' (bang bang), '!!!' (triple bang).
1045
- Again, we'll leave to Hardley the explanation on how to use all those functions.
2428
+ Again, we'll leave to Hadley the explanation on how to use all those functions.
1046
2429
 
1047
2430
  Now, let's try to implement the same function in galaaz. The next code block first prints the
1048
- 'df' data frame defined previously in R (to access an R variable from Galaaz, we use the tilda
1049
- operator '~' applied to the R variable name as symbol, i.e., ':df'.
2431
+ 'df' data frame defined previously in R (to access an R variable from Galaaz, we use the tilde
2432
+ operator `~` applied to the R variable name as a symbol, e.g. `:df`).
1050
2433
 
1051
2434
  ```{ruby r_dataframe}
1052
- puts ~:df
2435
+ puts ~R[:df]
1053
2436
  ```
1054
2437
 
1055
2438
  We then create the 'my_summarize' method and call it passing the R data frame and
1056
- the group by variable ':g1':
2439
+ the group by variable 'R[:g1]':
1057
2440
 
1058
2441
  ```{ruby diff_exp_ruby_func}
1059
2442
  def my_summarize(df, group_var)
1060
2443
  df.group_by(group_var).
1061
- summarize(a: :a.mean)
2444
+ summarize(a: R[:a].mean)
1062
2445
  end
1063
2446
 
1064
- puts my_summarize(:df, :g1).as__data__frame
2447
+ puts my_summarize(~R[:df], R[:g1])
1065
2448
  ```
1066
2449
 
1067
2450
  It works!!! Well, let's make sure this was not just some coincidence
1068
2451
 
1069
2452
  ```{ruby group_g2}
1070
- puts my_summarize(:df, :g2).as__data__frame
2453
+ puts my_summarize(~R[:df], R[:g2])
1071
2454
  ```
1072
2455
 
1073
2456
  Great, everything is fine! No magic, no new functions, no complexities, just normal, standard Ruby
@@ -1079,7 +2462,7 @@ In the previous section we've managed to get rid of all NSE formulation for a si
1079
2462
  does this remain true for more complex examples, or will the Galaaz way prove inpractical for
1080
2463
  more complex code?
1081
2464
 
1082
- In the next example Hardley proposes us to write a function that given an expression such as 'a'
2465
+ In the next example Hadley proposes us to write a function that given an expression such as 'a'
1083
2466
  or 'a * b', calculates three summaries. What we want a function that does the same as these R
1084
2467
  statements:
1085
2468
 
@@ -1108,9 +2491,9 @@ def my_summarise2(df, expr)
1108
2491
  )
1109
2492
  end
1110
2493
 
1111
- puts my_summarise2((~:df), :a)
2494
+ puts my_summarise2((~R[:df]), :a)
1112
2495
  puts "\n"
1113
- puts my_summarise2((~:df), :a * :b)
2496
+ puts my_summarise2((~R[:df]), R[:a] * R[:b])
1114
2497
  ```
1115
2498
 
1116
2499
  Once again, there is no need to use any special theory or functions. The only point to be
@@ -1118,7 +2501,7 @@ careful about is the use of 'E' to build expressions from functions 'mean', 'sum
1118
2501
 
1119
2502
  ## Different input and output variable
1120
2503
 
1121
- Now the next challenge presented by Hardley is to vary the name of the output variables based on
2504
+ Now the next challenge presented by Hadley is to vary the name of the output variables based on
1122
2505
  the received expression. So, if the input expression is 'a', we want our data frame columns to
1123
2506
  be named 'mean\_a' and 'sum\_a'. Now, if the input expression is 'b', columns
1124
2507
  should be named 'mean\_b' and 'sum\_b'.
@@ -1144,7 +2527,7 @@ mutate(df, mean_b = mean(b), sum_b = sum(b))
1144
2527
  #> 4 2 2 5 4 3 15
1145
2528
  #> # … with 1 more row
1146
2529
  ```
1147
- In order to solve this problem in R, Hardley needs to introduce some more new functions and notations:
2530
+ In order to solve this problem in R, Hadley needs to introduce some more new functions and notations:
1148
2531
  'quo_name' and the ':=' operator from package 'rlang'
1149
2532
 
1150
2533
  Here is our Ruby code:
@@ -1158,9 +2541,9 @@ def my_mutate(df, expr)
1158
2541
  sum_name => E.sum(expr))
1159
2542
  end
1160
2543
 
1161
- puts my_mutate((~:df), :a)
2544
+ puts my_mutate((~R[:df]), :a)
1162
2545
  puts "\n"
1163
- puts my_mutate((~:df), :b)
2546
+ puts my_mutate((~R[:df]), :b)
1164
2547
  ```
1165
2548
  It really seems that "Non Standard Evaluation" is actually quite standard in Galaaz! But, you
1166
2549
  might have noticed a small change in the way the arguments to the mutate method were called.
@@ -1172,7 +2555,7 @@ and variable mean\_name is not followed by ':' but by '=>'. This is standard Ru
1172
2555
 
1173
2556
  ## Capturing multiple variables
1174
2557
 
1175
- Moving on with new complexities, Hardley proposes us to solve the problem in which the
2558
+ Moving on with new complexities, Hadley proposes us to solve the problem in which the
1176
2559
  summarise function will receive any number of grouping variables.
1177
2560
 
1178
2561
  This again is quite standard Ruby. In order to receive an undefined number of paramenters
@@ -1184,7 +2567,7 @@ def my_summarise3(df, *group_vars)
1184
2567
  summarise(a: E.mean(:a))
1185
2568
  end
1186
2569
 
1187
- puts my_summarise3((~:df), :g1, :g2).as__data__frame
2570
+ puts my_summarise3((~R[:df]), R[:g1], R[:g2])
1188
2571
  ```
1189
2572
 
1190
2573
  ## Why does R require NSE and Galaaz does not?
@@ -1202,7 +2585,7 @@ In Ruby, there is no lazy evaluation of parameters and 'a' is always a variable
1202
2585
  Variables assume their value as soon as they are used, so 'x = a' is immediately evaluate and
1203
2586
  variable 'x' will receive the value of variable 'a' as soon as the Ruby statement is executed.
1204
2587
  Ruby also provides the notion of a symbol; ':a' is a symbol and does not evaluate to anything.
1205
- Galaaz uses Ruby symbols to build expressions that are not bound to anything: ':a.eq :b' is
2588
+ Galaaz uses Ruby symbols to build expressions that are not bound to anything: 'R[:a].eq R[:b]' is
1206
2589
  clearly an expression and has no relationship whatsoever with the statment 'a = b'. By using
1207
2590
  symbols, variables and expressions all the possible ambiguities that are found in R are
1208
2591
  eliminated in Galaaz.
@@ -1212,7 +2595,7 @@ of input they are expecting, they might be expecting regular variables or they m
1212
2595
  expecting expressions and the R function will know how to deal with an input of the form
1213
2596
  'a = b', now for the Ruby developer it might not be immediately clear if it should call the
1214
2597
  function passing the value 'true' if variable 'a' is equal to variable 'b' or if it should
1215
- call the function passing the expression ':a.eq :b'.
2598
+ call the function passing the expression 'R[:a].eq R[:b]'.
1216
2599
 
1217
2600
 
1218
2601
  ## Advanced dplyr features
@@ -1235,12 +2618,13 @@ In the following examples, we show the use of functions 'group\_by\_at', 'summar
1235
2618
  features of characters in the Starwars movies:
1236
2619
 
1237
2620
  ```{ruby starwars}
1238
- puts (~:starwars).head.as__data__frame
2621
+ puts (~R[:starwars]).head
1239
2622
  ```
1240
- The grouped_mean function bellow will receive a grouping variable and calculate summaries for
2623
+ The grouped_mean function below will receive a grouping variable and calculate summaries for
1241
2624
  the value\_variables given:
1242
2625
 
1243
2626
  ```{r grouped_mean}
2627
+ library(dplyr)
1244
2628
  grouped_mean <- function(data, grouping_variables, value_variables) {
1245
2629
  data %>%
1246
2630
  group_by_at(grouping_variables) %>%
@@ -1262,24 +2646,26 @@ def grouped_mean(data, grouping_variables, value_variables)
1262
2646
  data.
1263
2647
  group_by_at(grouping_variables).
1264
2648
  mutate(count: E.n).
1265
- summarise_at(E.c(value_variables, "count"), ~:mean, na__rm: true).
2649
+ summarise_at(E.c(value_variables, "count"), ~R[:mean], na__rm: true).
1266
2650
  rename_at(value_variables, E.funs(E.paste0("mean_", value_variables)))
1267
2651
  end
1268
2652
 
1269
- puts grouped_mean((~:starwars), "eye_color", E.c("mass", "birth_year")).as__data__frame
2653
+ puts grouped_mean((~R[:starwars]), "eye_color", E.c("mass", "birth_year"))
1270
2654
  ```
1271
2655
 
1272
-
1273
- [TO BE CONTINUED...]
1274
-
2656
+ The examples above cover programmatic dplyr with string column names and `_at` helpers. The same
2657
+ Galaaz patterns (symbols, `E.*` for expression-safe functions, and Ruby methods on R-backed objects)
2658
+ extend to other tidyverse workflows; consult R package documentation for function-specific
2659
+ arguments.
1275
2660
 
1276
2661
  # Contributing
1277
2662
 
1278
-
1279
2663
  * Fork it
1280
- * Create your feature branch (git checkout -b my-new-feature)
1281
- * Write Tests!
1282
- * Commit your changes (git commit -am 'Add some feature')
1283
- * Push to the branch (git push origin my-new-feature)
1284
- * Create new Pull Request
1285
-
2664
+ * Create your feature branch (`git checkout -b my-new-feature`)
2665
+ * Write tests — use **`bin/run_rspec`** or **`bin/run_all_rspec`** with **JRuby** so JVM flags and
2666
+ the load path match **`docs/testing.md`**
2667
+ * Commit your changes (`git commit -am 'Add some feature'`)
2668
+ * Push to the branch (`git push origin my-new-feature`)
2669
+ * Open a pull request
2670
+
2671
+ # References