galaaz 0.4.10 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +26 -0
- data/LICENSE +0 -0
- data/README.md +3123 -882
- data/Rakefile +62 -41
- data/bin/galaaz-bootstrap +137 -0
- data/bin/galaaz-jruby +14 -0
- data/bin/galaaz_jruby_env.inc.sh +6 -0
- data/bin/gbookdown +64 -0
- data/bin/gknit +223 -6
- data/bin/gknit-draft +105 -0
- data/bin/gknit-draft.rb +28 -0
- data/bin/gknit_Rscript +127 -0
- data/bin/grun +27 -1
- data/bin/gstudio +49 -4
- data/bin/{gstudio.rb → gstudio_irb.rb} +0 -0
- data/bin/gstudio_pry.rb +7 -0
- data/bin/install-tinytex +6 -0
- data/bin/run_all_rspec +43 -0
- data/bin/run_example +14 -0
- data/bin/run_old_rspec +19 -0
- data/bin/run_rspec +23 -0
- data/bin/run_rspec_subset +38 -0
- data/bin/run_slow_rspec +19 -0
- data/blogs/R-on-Rails-Planning-Document.md +940 -0
- data/blogs/README.md +100 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot.Rmd +38 -66
- data/blogs/galaaz_ggplot/galaaz_ggplot.log +754 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot.md +364 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot.tex +607 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-html/midwest_rb.png +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-html/scatter_plot_rb.png +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-markdown_github/midwest_rb.png +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-markdown_github/scatter_plot_rb.png +0 -0
- data/blogs/galaaz_ggplot/midwest.Rmd +3 -3
- data/blogs/galaaz_ggplot/midwest_external_png +0 -0
- data/blogs/gknit/gknit.Rmd +52 -55
- data/blogs/gknit/gknit.md +94 -94
- data/blogs/gknit/gknit_files/figure-html/bubble-1.png +0 -0
- data/blogs/gknit/gknit_files/figure-html/diverging_bar.png +0 -0
- data/blogs/gknit/lst.rds +0 -0
- data/blogs/gknit/model.rb +1 -1
- data/blogs/gknit/stats.bib +0 -0
- data/blogs/manual/include_model_local_repro.Rmd +14 -0
- data/blogs/manual/include_model_local_repro.md +75 -0
- data/blogs/manual/lst.rds +0 -0
- data/blogs/manual/manual.Rmd +1582 -196
- data/blogs/manual/manual.log +1786 -0
- data/blogs/manual/manual.md +3107 -890
- data/blogs/manual/manual.tex +3018 -1086
- data/blogs/manual/manual_files/figure-html/bubble-1.png +0 -0
- data/blogs/manual/manual_files/figure-html/diverging_bar.png +0 -0
- data/blogs/manual/manual_files/figure-latex/bubble-1.png +0 -0
- data/blogs/manual/model.rb +41 -0
- data/blogs/nse_dplyr/nse_dplyr.Rmd +277 -151
- data/blogs/nse_dplyr/nse_dplyr.log +928 -0
- data/blogs/nse_dplyr/nse_dplyr.md +457 -293
- data/blogs/oh_my/not_so.rb +0 -0
- data/blogs/oh_my/oh_my.Rmd +1234 -25
- data/blogs/oh_my/oh_my.log +804 -0
- data/blogs/oh_my/oh_my.md +1808 -228
- data/blogs/oh_my/oh_my.tex +821 -0
- data/blogs/oh_my/old.Rmd +15 -14
- data/blogs/ruby_plot/ruby_plot.Rmd +58 -82
- data/blogs/ruby_plot/ruby_plot.log +885 -0
- data/blogs/ruby_plot/ruby_plot.md +71 -103
- data/blogs/ruby_plot/ruby_plot.tex +940 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/dose_len.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_delivery.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_dose.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color2.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_decorations.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_points.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/final_box_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/final_violin_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/violin_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/dose_len.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facet_by_delivery.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facet_by_dose.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_by_delivery_color.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_by_delivery_color2.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_decorations.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_points.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/final_box_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/final_violin_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/violin_with_jitter.png +0 -0
- data/blogs/test/test.Rmd +14 -0
- data/examples/50Plots_MasterList/Images/midwest-scatterplot.PNG +0 -0
- data/examples/50Plots_MasterList/ScatterPlot.rb +0 -0
- data/examples/50Plots_MasterList/scatter_plot.rb +0 -0
- data/examples/Bibliography/master.bib +50 -0
- data/examples/Bibliography/stats.bib +72 -0
- data/examples/R/calc.R +0 -0
- data/examples/R/java_interop.R +0 -0
- data/examples/bioconductor_deseq2_airway/Documentation/DESeq2-airway-walkthrough.md +56 -0
- data/examples/bioconductor_deseq2_airway/bench_galaaz_three_same_process.rb +53 -0
- data/examples/bioconductor_deseq2_airway/bench_r_three_same_process.R +34 -0
- data/examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb +33 -0
- data/examples/bioconductor_deseq2_airway/deseq2_airway_galaaz_optimized.rb +34 -0
- data/examples/bioconductor_deseq2_airway/deseq2_airway_minimal.R +30 -0
- data/examples/bioconductor_deseq2_airway/deseq2_airway_pipeline_for_bench.R +36 -0
- data/examples/islr/all.rb +13 -0
- data/examples/islr/ch2.spec.rb +37 -7
- data/examples/islr/ch3.spec.rb +11 -2
- data/examples/islr/ch3_boston.rb +27 -0
- data/examples/islr/ch3_multiple_regression.rb +0 -0
- data/examples/islr/ch6.spec.rb +24 -1
- data/examples/islr/x_y_rnorm.jpg +0 -0
- data/examples/latex_templates/Test-acm_article/Makefile +16 -0
- data/examples/latex_templates/Test-acm_article/Test-acm_article.Rmd +65 -0
- data/examples/latex_templates/Test-acm_article/acm_proc_article-sp.cls +1670 -0
- data/examples/latex_templates/Test-acm_article/sensys-abstract.cls +703 -0
- data/examples/latex_templates/Test-acm_article/sigproc.bib +59 -0
- data/examples/latex_templates/Test-acs_article/Test-acs_article.Rmd +260 -0
- data/examples/latex_templates/Test-acs_article/acs-Test-acs_article.bib +11 -0
- data/examples/latex_templates/Test-acs_article/acs-my_output.bib +11 -0
- data/examples/latex_templates/Test-acs_article/acstest.bib +17 -0
- data/examples/latex_templates/Test-aea_article/AEA.cls +1414 -0
- data/{blogs/gknit/marshal.dump → examples/latex_templates/Test-aea_article/BibFile.bib} +0 -0
- data/examples/latex_templates/Test-aea_article/Test-aea_article.Rmd +108 -0
- data/examples/latex_templates/Test-aea_article/aea.bst +1269 -0
- data/examples/latex_templates/Test-aea_article/multicol.sty +853 -0
- data/examples/latex_templates/Test-aea_article/references.bib +0 -0
- data/examples/latex_templates/Test-aea_article/setspace.sty +546 -0
- data/examples/latex_templates/Test-amq_article/Test-amq_article.Rmd +256 -0
- data/examples/latex_templates/Test-amq_article/Test-amq_article.pdfsync +3397 -0
- data/examples/latex_templates/Test-ams_article/Test-ams_article.Rmd +215 -0
- data/examples/latex_templates/Test-ams_article/amstest.bib +436 -0
- data/examples/latex_templates/Test-asa_article/Test-asa_article.Rmd +153 -0
- data/examples/latex_templates/Test-asa_article/agsm.bst +1353 -0
- data/examples/latex_templates/Test-asa_article/bibliography.bib +233 -0
- data/examples/latex_templates/Test-ieee_article/IEEEtran.bst +2409 -0
- data/examples/latex_templates/Test-ieee_article/IEEEtran.cls +6346 -0
- data/examples/latex_templates/Test-ieee_article/Test-ieee_article.Rmd +175 -0
- data/examples/latex_templates/Test-ieee_article/mybibfile.bib +20 -0
- data/examples/latex_templates/Test-rjournal_article/RJournal.sty +335 -0
- data/examples/latex_templates/Test-rjournal_article/RJreferences.bib +18 -0
- data/examples/latex_templates/Test-rjournal_article/Test-rjournal_article.Rmd +52 -0
- data/examples/latex_templates/Test-springer_article/Test-springer_article.Rmd +65 -0
- data/examples/latex_templates/Test-springer_article/bibliography.bib +26 -0
- data/examples/latex_templates/Test-springer_article/spbasic.bst +1658 -0
- data/examples/latex_templates/Test-springer_article/spmpsci.bst +1512 -0
- data/examples/latex_templates/Test-springer_article/spphys.bst +1443 -0
- data/examples/latex_templates/Test-springer_article/svglov3.clo +113 -0
- data/examples/latex_templates/Test-springer_article/svjour3.cls +1431 -0
- data/examples/misc/baseball.csv +0 -0
- data/examples/misc/ggplot.rb +3 -2
- data/examples/misc/moneyball.rb +0 -0
- data/examples/misc/subsetting.rb +0 -0
- data/examples/multithread_shards_to_r/shards_to_r.rb +67 -0
- data/examples/rmarkdown/svm-rmarkdown-anon-ms-example/svm-rmarkdown-anon-ms-example.Rmd +73 -0
- data/examples/rmarkdown/svm-rmarkdown-article-example/svm-rmarkdown-article-example.Rmd +382 -0
- data/examples/rmarkdown/svm-rmarkdown-beamer-example/svm-rmarkdown-beamer-example.Rmd +164 -0
- data/examples/rmarkdown/svm-rmarkdown-cv/svm-rmarkdown-cv.Rmd +92 -0
- data/examples/rmarkdown/svm-rmarkdown-syllabus-example/attend-grade-relationships.csv +482 -0
- data/examples/rmarkdown/svm-rmarkdown-syllabus-example/svm-rmarkdown-syllabus-example.Rmd +280 -0
- data/examples/rmarkdown/svm-xaringan-example/svm-xaringan-example.Rmd +386 -0
- data/examples/sthda_ggplot/README.md +0 -0
- data/examples/sthda_ggplot/RUN.md +41 -0
- data/examples/sthda_ggplot/all.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/density_gg.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_area.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_density.rb +2 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_dotplot.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_freqpoly.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_histogram.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/histogram_density.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/stat.rb +0 -0
- data/examples/sthda_ggplot/one_variable_discrete/bar.rb +0 -0
- data/examples/sthda_ggplot/qplots/box_violin_dot.rb +0 -0
- data/examples/sthda_ggplot/qplots/scatter_plots.rb +0 -0
- data/examples/sthda_ggplot/scatter_gg.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_bivariate/geom_bin2d.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_bivariate/geom_density2d.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_bivariate/geom_hex.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_cont/geom_point.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_cont/geom_smooth.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_cont/misc.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_function/geom_area.rb +4 -3
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_bar.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_boxplot.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_dotplot.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_jitter.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_line.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_violin.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_disc/geom_jitter.rb +0 -0
- data/examples/sthda_ggplot/two_variables_error/geom_crossbar.rb +0 -0
- data/ext/new_bridge/Makefile +46 -0
- data/ext/new_bridge/galaaz_gatekeeper_phase0.cpp +12 -0
- data/ext/new_bridge/galaaz_gatekeeper_phase1.cpp +1639 -0
- data/lib/R_interface/galaaz_device.R +20 -0
- data/lib/R_interface/include_engine.R +109 -0
- data/lib/R_interface/new_bridge_adapter.rb +824 -0
- data/lib/R_interface/r.rb +177 -25
- data/lib/R_interface/r_arrow.rb +113 -0
- data/lib/R_interface/r_libs.R +4 -4
- data/lib/R_interface/r_methods.rb +13 -116
- data/lib/R_interface/r_module_s.rb +0 -0
- data/lib/R_interface/rbinary_operators.rb +20 -2
- data/lib/R_interface/rclosure.rb +5 -1
- data/lib/R_interface/rdata_frame.rb +34 -70
- data/lib/R_interface/rdevice.rb +125 -0
- data/lib/R_interface/rdevices.R +0 -0
- data/lib/R_interface/renvironment.rb +10 -4
- data/lib/R_interface/rexpression.rb +5 -1
- data/lib/R_interface/rindexed_object.rb +41 -13
- data/lib/R_interface/rlanguage.rb +20 -62
- data/lib/R_interface/rlist.rb +115 -25
- data/lib/R_interface/rlogical_operators.rb +0 -0
- data/lib/R_interface/rmatrix.rb +2 -11
- data/lib/R_interface/rmd_indexed_object.rb +5 -1
- data/lib/R_interface/robject.rb +348 -290
- data/lib/R_interface/rpkg.rb +1 -0
- data/lib/R_interface/rsupport.rb +610 -331
- data/lib/R_interface/rsupport_scope.rb +2 -1
- data/lib/R_interface/rsymbol.rb +50 -0
- data/lib/R_interface/ruby_callback.rb +2 -3
- data/lib/R_interface/ruby_extensions.rb +225 -175
- data/lib/R_interface/runary_operators.rb +0 -0
- data/lib/R_interface/rvector.rb +147 -31
- data/lib/galaaz.rb +0 -0
- data/lib/galaaz_jruby.rb +22 -0
- data/lib/gknit/diagnostics.rb +50 -0
- data/lib/gknit/draft.rb +111 -0
- data/lib/gknit/include_engine.rb +15 -7
- data/lib/gknit/knitr_engine.rb +223 -107
- data/lib/gknit/rb_engine.rb +3 -3
- data/lib/gknit/ruby_engine.rb +0 -0
- data/lib/gknit.rb +3 -0
- data/lib/new_bridge/bootstrap/windows_bootstrap.rb +285 -0
- data/lib/new_bridge/envelope.rb +51 -0
- data/lib/new_bridge/eval_result.rb +26 -0
- data/lib/new_bridge/framing.rb +39 -0
- data/lib/new_bridge/instance_pool_client.rb +38 -0
- data/lib/new_bridge/r_instance_manager.rb +404 -0
- data/lib/new_bridge/session_client.rb +530 -0
- data/lib/new_bridge/tcp_framed.rb +44 -0
- data/lib/new_bridge.rb +9 -0
- data/lib/util/exec_ruby.rb +95 -46
- data/lib/util/inline_file.rb +35 -30
- data/new_bridge_specs/benchmark_phase5_5_unboxing_spec.rb +96 -0
- data/new_bridge_specs/eval_r_async_spec.rb +113 -0
- data/new_bridge_specs/integration_phase5_1_concurrent_spec.rb +50 -0
- data/new_bridge_specs/integration_phase5_1_eval_spec.rb +16 -0
- data/new_bridge_specs/integration_phase5_1_r_api_spec.rb +25 -0
- data/new_bridge_specs/integration_phase5_1_smoke_spec.rb +31 -0
- data/new_bridge_specs/integration_phase5_2_dataframe_unboxing_spec.rb +19 -0
- data/new_bridge_specs/integration_phase5_2_handle_eval_unboxing_spec.rb +25 -0
- data/new_bridge_specs/integration_phase5_3_callback_args_spec.rb +28 -0
- data/new_bridge_specs/integration_phase5_3_callback_error_spec.rb +22 -0
- data/new_bridge_specs/integration_phase5_3_callback_timeout_spec.rb +28 -0
- data/new_bridge_specs/integration_phase5_3_callbacks_smoke_spec.rb +22 -0
- data/new_bridge_specs/integration_phase5_3_edge_cases_spec.rb +52 -0
- data/new_bridge_specs/integration_phase5_3_nested_spec.rb +30 -0
- data/new_bridge_specs/integration_phase5_4_concurrent_sessions_spec.rb +53 -0
- data/new_bridge_specs/integration_phase5_4_nested_session_callbacks_spec.rb +49 -0
- data/new_bridge_specs/integration_phase5_4_session_routing_spec.rb +38 -0
- data/new_bridge_specs/integration_phase5_5_stress_concurrency_spec.rb +52 -0
- data/new_bridge_specs/integration_phase5_5_unbox_walk_spec.rb +46 -0
- data/new_bridge_specs/phase0_protocol_spec.rb +96 -0
- data/new_bridge_specs/phase1_req_ret_spec.rb +66 -0
- data/new_bridge_specs/phase2_multi_instance_spec.rb +67 -0
- data/new_bridge_specs/phase3_callbacks_spec.rb +71 -0
- data/new_bridge_specs/phase4_2_hardening_spec.rb +252 -0
- data/new_bridge_specs/phase4_3_r_instance_manager_spec.rb +85 -0
- data/new_bridge_specs/phase4_nested_callbacks_spec.rb +123 -0
- data/r_requires/ggplot.rb +0 -0
- data/r_requires/knitr.rb +0 -0
- data/specs/all.rb +15 -11
- data/specs/arrow_from_ruby_batches_spec.rb +50 -0
- data/specs/arrow_semantics_spec.rb +64 -0
- data/specs/bridge_concurrent_spec.rb +46 -0
- data/specs/bridge_nested_spec.rb +25 -0
- data/specs/dataframe_semantics_spec.rb +122 -0
- data/specs/dataframe_single_index_logical_filter_spec.rb +21 -0
- data/specs/dispatch_probe_cache_spec.rb +38 -0
- data/specs/dispatch_probe_error_class_fallback_spec.rb +20 -0
- data/specs/dispatch_probe_fallback_spec.rb +18 -0
- data/specs/environment_semantics_spec.rb +89 -0
- data/specs/field_access_spec.rb +31 -0
- data/specs/figures/bg.jpeg +0 -0
- data/specs/figures/bg.png +0 -0
- data/specs/figures/bg.svg +168 -57
- data/specs/figures/dose_len.png +0 -0
- data/specs/figures/no_args.jpeg +0 -0
- data/specs/figures/no_args.png +0 -0
- data/specs/figures/no_args.svg +168 -57
- data/specs/figures/width_height.jpeg +0 -0
- data/specs/figures/width_height.png +0 -0
- data/specs/figures/width_height_units1.jpeg +0 -0
- data/specs/figures/width_height_units1.png +0 -0
- data/specs/figures/width_height_units2.jpeg +0 -0
- data/specs/figures/width_height_units2.png +0 -0
- data/specs/formula_semantics_spec.rb +81 -0
- data/specs/galaaz_util_exec_ruby_spec.rb +85 -0
- data/specs/galaaz_util_inline_file_spec.rb +54 -0
- data/specs/gknit_cli_option_permutation_spec.rb +24 -0
- data/specs/gknit_include_engine_spec.rb +72 -0
- data/specs/gknit_install_timeout_report_spec.rb +69 -0
- data/specs/gknit_internal_error_report_spec.rb +57 -0
- data/specs/gknit_vector_map_output_spec.rb +59 -0
- data/specs/globalenv_guardrail_spec.rb +52 -0
- data/specs/language_expression_semantics_spec.rb +145 -0
- data/specs/list_semantics_spec.rb +111 -0
- data/specs/new_bridge_bulk_dataframe_transfer_spec.rb +44 -0
- data/specs/new_bridge_bulk_vector_transfer_spec.rb +73 -0
- data/specs/new_bridge_callback_timeout_spec.rb +69 -0
- data/specs/new_bridge_eval_r_fallback_spec.rb +55 -0
- data/specs/nil_null_spec.rb +42 -0
- data/specs/object_build_phase2_spec.rb +53 -0
- data/specs/phase1_callback_bridge_spec.rb +84 -0
- data/specs/phase2_gknit_generic_rendering_guardrail_spec.rb +46 -0
- data/specs/phase2_gknit_no_raw_code_leakage_spec.rb +43 -0
- data/specs/phase3_gknit_generic_graphics_capture_spec.rb +71 -0
- data/specs/plot_device_semantics_spec.rb +28 -0
- data/specs/plot_snapshot_semantics_spec.rb +58 -0
- data/specs/protocol_result_spec.rb +236 -0
- data/specs/r_batch_fail_fast_spec.rb +47 -0
- data/specs/r_bridge_bootstrap_spec.rb +11 -0
- data/specs/r_devices.spec.rb +1 -1
- data/specs/r_eval.spec.rb +16 -18
- data/specs/r_function.spec.rb +1 -1
- data/specs/r_instance_manager_spec.rb +285 -0
- data/specs/r_list_apply.spec.rb +15 -15
- data/specs/r_matrix.spec.rb +0 -0
- data/specs/r_nse.spec.rb +5 -5
- data/specs/r_object_send_dispatch_spec.rb +13 -0
- data/specs/r_vector_comparator_spec.rb +8 -0
- data/specs/r_vector_creation.spec.rb +0 -0
- data/specs/r_vector_functions.spec.rb +0 -0
- data/specs/r_vector_object.spec.rb +0 -0
- data/specs/r_vector_operators.spec.rb +0 -0
- data/specs/r_vector_structured_scalar_reads_spec.rb +35 -0
- data/specs/r_vector_subsetting.spec.rb +0 -0
- data/specs/range_helper_spec.rb +21 -0
- data/specs/rsupport_scope_spec.rb +28 -0
- data/specs/rsupport_var_name_thread_safety_spec.rb +24 -0
- data/specs/scalar_character_spec.rb +44 -0
- data/specs/scoped_symbol_dsl_refinement_spec.rb +40 -0
- data/specs/session_env_bridge_spec.rb +25 -0
- data/specs/simplecov_bootstrap_spec.rb +10 -0
- data/specs/spec_helper.rb +10 -0
- data/specs/tmp.rb +41 -20
- data/specs/unboxing_recursion_regression_spec.rb +30 -0
- data/specs/unboxing_spec.rb +49 -0
- data/specs/verify_callbacks.rb +42 -0
- data/sty/galaaz.sty +0 -0
- data/version.rb +1 -1
- metadata +239 -71
- data/blogs/galaaz_ggplot/galaaz_ggplot.aux +0 -41
- data/blogs/galaaz_ggplot/galaaz_ggplot.html +0 -705
- data/blogs/galaaz_ggplot/galaaz_ggplot.out +0 -10
- data/blogs/galaaz_ggplot/galaaz_ggplot.pdf +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-latex/midwest_rb.pdf +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-latex/scatter_plot_rb.pdf +0 -0
- data/blogs/galaaz_ggplot/midwest.html +0 -188
- data/blogs/gknit/gknit.html +0 -2266
- data/blogs/gknit/gknit.pdf +0 -0
- data/blogs/gknit/gknit.tex +0 -1358
- data/blogs/manual/graph.rb +0 -29
- data/blogs/manual/manual.html +0 -2995
- data/blogs/manual/manual.pdf +0 -0
- data/blogs/manual/manual_files/figure-latex/diverging_bar.pdf +0 -0
- data/blogs/nse_dplyr/nse_dplyr.html +0 -960
- data/blogs/nse_dplyr/nse_dplyr.pdf +0 -0
- data/blogs/nse_dplyr/nse_dplyr.tex +0 -1373
- data/blogs/oh_my/oh_my.html +0 -680
- data/blogs/ruby_plot/ruby_plot.Rmd_external_figs +0 -662
- data/blogs/ruby_plot/ruby_plot.html +0 -729
- data/blogs/ruby_plot/ruby_plot.pdf +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/dose_len.svg +0 -57
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_delivery.svg +0 -106
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_dose.svg +0 -110
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color.svg +0 -174
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color2.svg +0 -236
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_jitter.svg +0 -296
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_points.svg +0 -236
- data/blogs/ruby_plot/ruby_plot_files/figure-html/final_box_plot.svg +0 -218
- data/blogs/ruby_plot/ruby_plot_files/figure-html/final_violin_plot.svg +0 -128
- data/blogs/ruby_plot/ruby_plot_files/figure-html/violin_with_jitter.svg +0 -150
- data/examples/paper/paper.rb +0 -36
- data/specs/r_dataframe.spec.rb +0 -379
- data/specs/r_environment.spec.rb +0 -140
- data/specs/r_formula.spec.rb +0 -232
- data/specs/r_language.spec.rb +0 -112
- data/specs/r_list.spec.rb +0 -293
- data/specs/r_plots.spec.rb +0 -72
- data/specs/ruby_expression.spec.rb +0 -315
data/blogs/manual/manual.Rmd
CHANGED
|
@@ -1,26 +1,34 @@
|
|
|
1
1
|
---
|
|
2
2
|
title: "Galaaz Manual"
|
|
3
|
-
subtitle: "
|
|
3
|
+
subtitle: "Coupling Ruby (JRuby) and GNU R for data science"
|
|
4
4
|
author: "Rodrigo Botafogo"
|
|
5
|
-
tags: [Galaaz, Ruby, R,
|
|
6
|
-
date: "
|
|
5
|
+
tags: [Galaaz, Ruby, JRuby, R, "GNU R", ggplot2, knitr, dplyr, Bioconductor, Arrow]
|
|
6
|
+
date: "2026"
|
|
7
|
+
bibliography: "../../examples/Bibliography/stats.bib"
|
|
7
8
|
output:
|
|
9
|
+
html_document:
|
|
10
|
+
self_contained: true
|
|
11
|
+
keep_md: true
|
|
12
|
+
toc: true
|
|
13
|
+
toc_depth: 3
|
|
14
|
+
number_sections: true
|
|
8
15
|
pdf_document:
|
|
9
16
|
includes:
|
|
10
17
|
in_header: "../../sty/galaaz.sty"
|
|
11
18
|
keep_tex: yes
|
|
12
19
|
number_sections: yes
|
|
13
20
|
toc: true
|
|
14
|
-
toc_depth:
|
|
15
|
-
html_document:
|
|
16
|
-
self_contained: true
|
|
17
|
-
keep_md: true
|
|
21
|
+
toc_depth: 3
|
|
18
22
|
md_document:
|
|
19
23
|
variant: markdown_github
|
|
20
24
|
fontsize: 11pt
|
|
21
25
|
---
|
|
22
26
|
|
|
23
27
|
```{ruby setup, echo=FALSE}
|
|
28
|
+
# Bridge default is 60s; some chunks (Arrow, large dplyr pipes) need more.
|
|
29
|
+
ENV['GALAAZ_BRIDGE_TIMEOUT_SEC'] ||= '300'
|
|
30
|
+
|
|
31
|
+
R.options(crayon__enabled: false)
|
|
24
32
|
R.install_and_loads('kableExtra')
|
|
25
33
|
```
|
|
26
34
|
|
|
@@ -30,30 +38,272 @@ Galaaz is a system for tightly coupling Ruby and R. Ruby is a powerful language,
|
|
|
30
38
|
community, a very large set of libraries and great for web development. However, it lacks
|
|
31
39
|
libraries for data science, statistics, scientific plotting and machine learning. On the
|
|
32
40
|
other hand, R is considered one of the most powerful languages for solving all of the above
|
|
33
|
-
problems.
|
|
34
|
-
|
|
41
|
+
problems. **Python** is a strong competitor: NumPy, pandas, SciPy, and scikit-learn are
|
|
42
|
+
widely used building blocks, and **PyPI** hosts many thousands of other packages for
|
|
43
|
+
numerical work, machine learning, and beyond.
|
|
44
|
+
|
|
45
|
+
With Galaaz we do not intend to re-implement any of the scientific libraries in R, we allow
|
|
46
|
+
for very tight coupling between the two languages to the point that the Ruby developer does
|
|
47
|
+
not need to know that there is an R engine running.
|
|
48
|
+
|
|
49
|
+
According to Wikipedia "Ruby is a dynamic, interpreted, reflective, object-oriented,
|
|
50
|
+
general-purpose programming language. It was designed and developed in the mid-1990s by Yukihiro
|
|
51
|
+
"Matz" Matsumoto in Japan." It reached high popularity with the development of Ruby on Rails
|
|
52
|
+
(RoR) by David Heinemeier Hansson. RoR is a web application framework first released
|
|
53
|
+
around 2005. It makes extensive use of Ruby's metaprogramming features. With RoR,
|
|
54
|
+
Ruby became very popular. According to [Ruby’s place in the TIOBE index](https://www.tiobe.com/tiobe-index/ruby/)
|
|
55
|
+
it peaked in popularity around 2008, then declined until 2015 when it started picking up again.
|
|
56
|
+
Ruby remains a significant language in web development and general-purpose scripting.
|
|
57
|
+
|
|
58
|
+
Python, a language similar to Ruby, ranks 4th in the index. Java, C and C++ take the
|
|
59
|
+
first three positions. Ruby is often criticized for its focus on web applications.
|
|
60
|
+
But Ruby can do [much more](https://github.com/markets/awesome-ruby) than just web applications.
|
|
61
|
+
Yet, for scientific computing, Ruby lags behind Python and R. Python offers Django and
|
|
62
|
+
similar frameworks for the web, plus NumPy, pandas, and a deep catalog of science and ML libraries.
|
|
63
|
+
R is a free software environment for statistical computing and graphics with thousands
|
|
64
|
+
of libraries for data analysis.
|
|
65
|
+
|
|
66
|
+
Until recently, there was no real perspective for Ruby to bridge this gap.
|
|
67
|
+
Implementing a complete scientific computing infrastructure would take too long.
|
|
68
|
+
|
|
69
|
+
**Galaaz 2.0** couples **JRuby** (Ruby on the JVM) with **GNU R**—the same R you use for
|
|
70
|
+
CRAN and Bioconductor. Ruby and R run in **separate processes**; the **Galaaz bridge**
|
|
71
|
+
sends requests to R and returns results to Ruby. From your point of view you still write
|
|
72
|
+
Ruby: `R.c(...)`, `R.library('ggplot2')`, `~R[:mtcars]`, and dplyr-style chains on R objects.
|
|
73
|
+
You do not need to learn R syntax to get a lot done, though reading R documentation for
|
|
74
|
+
individual packages remains useful.
|
|
75
|
+
|
|
76
|
+
Earlier experiments with Galaaz used Oracle’s **GraalVM** with TruffleRuby and FastR so that
|
|
77
|
+
Ruby and R could share one runtime. That path is no longer the focus: **standard GNU R**
|
|
78
|
+
gives full compatibility with the R package ecosystem (including compiled extensions and
|
|
79
|
+
Bioconductor) while JRuby gives a mature Ruby with **real multithreading** for application
|
|
80
|
+
and I/O code.
|
|
81
|
+
|
|
82
|
+
The bridge handles **communication and typing** between the two worlds; large tables can
|
|
83
|
+
also flow through **Apache Arrow** on the R side when you use the optional helpers described
|
|
84
|
+
later in this manual.
|
|
85
|
+
|
|
86
|
+
Library wrapping is a common way to bring features from one language into another.
|
|
87
|
+
To improve performance, Python often wraps more efficient C libraries. For the
|
|
88
|
+
Python developer, the existence of such C libraries is hidden. The problem with
|
|
89
|
+
library wrapping is that for any new library, there is the need to handcraft a new
|
|
90
|
+
wrapper.
|
|
91
|
+
|
|
92
|
+
Galaaz, instead of wrapping a single C or R library, wraps the whole R language
|
|
93
|
+
in Ruby. Doing so, all thousands of R libraries are available immediately
|
|
94
|
+
to Ruby developers without any new wrapping effort.
|
|
95
|
+
|
|
96
|
+
## What does Galaaz mean
|
|
97
|
+
|
|
98
|
+
Galaaz is the Portuguese name for "Galahad". From Wikipedia:
|
|
99
|
+
|
|
100
|
+
Sir Galahad (sometimes referred to as Galeas or Galath),
|
|
101
|
+
in Arthurian legend, is a knight of King Arthur's Round Table and one
|
|
102
|
+
of the three achievers of the Holy Grail. He is the illegitimate son
|
|
103
|
+
of Sir Lancelot and Elaine of Corbenic, and is renowned for his
|
|
104
|
+
gallantry and purity as the most perfect of all knights. Emerging quite
|
|
105
|
+
late in the medieval Arthurian tradition, Sir Galahad first appears in the
|
|
106
|
+
Lancelot–Grail cycle, and his story is taken up in later works such as
|
|
107
|
+
the Post-Vulgate Cycle and Sir Thomas Malory's Le Morte d'Arthur.
|
|
108
|
+
His name should not be mistaken with Galehaut, a different knight from
|
|
109
|
+
Arthurian legend.
|
|
110
|
+
|
|
111
|
+
# Command-line tools (`bin/`)
|
|
112
|
+
|
|
113
|
+
The Galaaz repository ships many helpers under **`bin/`**. When working from a **clone**, call
|
|
114
|
+
them as **`bin/<name>`** from the project root (or `./bin/<name>`). If you install the **gem**,
|
|
115
|
+
only a subset is guaranteed on your `PATH` (see the gemspec: **`galaaz`**, **`gstudio`**, **`gknit`**, **`grun`**, **`gknit-draft`**); for development and CI, prefer the **`bin/`** copies so JVM flags and paths stay correct.
|
|
116
|
+
|
|
117
|
+
Below, **current (Galaaz 2.0 + JRuby + GNU R)** means the tool is wired to **`jruby`** and
|
|
118
|
+
**`bin/galaaz_jruby_env.inc.sh`** (or equivalent logic in Ruby via `lib/galaaz_jruby.rb`). **Legacy**
|
|
119
|
+
means the script still targets **GraalVM** polyglot Ruby / FastR-era invocation and is **not**
|
|
120
|
+
expected to work on a typical JRuby-only setup.
|
|
121
|
+
|
|
122
|
+
**Table layout:** names in the first column are **`bin/`** filenames (run as `bin/<name>` from the repo root). Long options and examples sit **outside** the tables so PDF columns stay readable.
|
|
123
|
+
|
|
124
|
+
```{r bin-tables-helper, echo=FALSE}
|
|
125
|
+
bin_tbl <- function(df) {
|
|
126
|
+
k <- knitr::kable(df, row.names = FALSE, booktabs = TRUE, linesep = "",
|
|
127
|
+
col.names = c("Script", "Role", "2.0?"))
|
|
128
|
+
if (knitr::is_latex_output()) {
|
|
129
|
+
k <- kableExtra::kable_styling(k, font_size = 9, latex_options = "scale_down")
|
|
130
|
+
k <- kableExtra::column_spec(k, 1, width = "2.5cm")
|
|
131
|
+
k <- kableExtra::column_spec(k, 2, width = "9.5cm")
|
|
132
|
+
k <- kableExtra::column_spec(k, 3, width = "2.8cm")
|
|
133
|
+
} else {
|
|
134
|
+
k <- kableExtra::kable_styling(k, bootstrap_options = c("striped", "condensed"), full_width = TRUE)
|
|
135
|
+
}
|
|
136
|
+
k
|
|
137
|
+
}
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
```{r bin-tables-bootstrap, echo=FALSE}
|
|
141
|
+
df_boot <- data.frame(
|
|
142
|
+
Script = c("galaaz-bootstrap", "galaaz-jruby", "galaaz_jruby_env.inc.sh", "install-tinytex"),
|
|
143
|
+
Role = c(
|
|
144
|
+
"WSL2 helper: Docker checks; optional TinyTeX or poppler for gKnit PDF.",
|
|
145
|
+
"JRuby with repo lib/ on LOAD_PATH and required JVM flags (e.g. Arrow).",
|
|
146
|
+
"Sourced by bash wrappers; sets GALAAZ_REQUIRED_JRUBY_J_ARGS.",
|
|
147
|
+
"Install TinyTeX for PDF output."
|
|
148
|
+
),
|
|
149
|
+
X2 = c("Yes*", "Yes", "Yes†", "Yes"),
|
|
150
|
+
stringsAsFactors = FALSE
|
|
151
|
+
)
|
|
152
|
+
bin_tbl(df_boot)
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
\* Where WSL/Docker apply. **`galaaz-bootstrap` flags:** `--check`, `--apply`, `--runtime` (`docker` \| `local` \| `auto`), `--[no-]prompt-doc-tools`.
|
|
156
|
+
|
|
157
|
+
† Not run directly.
|
|
158
|
+
|
|
159
|
+
**`galaaz-jruby` examples** (from repo root):
|
|
160
|
+
|
|
161
|
+
```text
|
|
162
|
+
bin/galaaz-jruby my_script.rb
|
|
163
|
+
bin/galaaz-jruby -S rspec
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## Interactive use, examples, and Rake
|
|
167
|
+
|
|
168
|
+
```{r bin-tables-interactive, echo=FALSE}
|
|
169
|
+
df_ix <- data.frame(
|
|
170
|
+
Script = c("gstudio", "run_example", "galaaz"),
|
|
171
|
+
Role = c(
|
|
172
|
+
"IRB or Pry with Galaaz preloaded (JRuby + JVM flags).",
|
|
173
|
+
"Run one Ruby file using the same JRuby/JVM setup as tests.",
|
|
174
|
+
"Forward arguments to rake (needs rake; usually JRuby)."
|
|
175
|
+
),
|
|
176
|
+
X2 = c("Yes", "Yes", "Yes"),
|
|
177
|
+
stringsAsFactors = FALSE
|
|
178
|
+
)
|
|
179
|
+
bin_tbl(df_ix)
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
## gKnit and document drafts
|
|
183
|
+
|
|
184
|
+
```{r bin-tables-gknit, echo=FALSE}
|
|
185
|
+
df_gk <- data.frame(
|
|
186
|
+
Script = c("gknit", "gknit-draft", "gknit-draft.rb", "gknit_Rscript"),
|
|
187
|
+
Role = c(
|
|
188
|
+
"Knit .Rmd via JRuby and R Markdown render.",
|
|
189
|
+
"Drafts from rticles-style templates; wrapper still uses legacy polyglot ruby.",
|
|
190
|
+
"Ruby entry: GKnit.draft (use with JRuby + LOAD_PATH).",
|
|
191
|
+
"Polyglot Rscript launcher; hard-coded LOAD_PATH sample."
|
|
192
|
+
),
|
|
193
|
+
X2 = c("Yes", "Legacy", "JRuby", "No"),
|
|
194
|
+
stringsAsFactors = FALSE
|
|
195
|
+
)
|
|
196
|
+
bin_tbl(df_gk)
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
**`gknit` CLI** (see `gknit -h`): `--output_format`, `--output_file`, `--output_dir`, `--bridge_timeout_sec`, `--callback_timeout_ms`. If `--output_format` is omitted, the **first** YAML `output:` target wins.
|
|
200
|
+
|
|
201
|
+
Prefer **`galaaz-jruby`** for **`gknit-draft`** workflows until that wrapper matches the **`gknit`** stack.
|
|
202
|
+
|
|
203
|
+
## Tests
|
|
204
|
+
|
|
205
|
+
```{r bin-tables-tests, echo=FALSE}
|
|
206
|
+
df_ts <- data.frame(
|
|
207
|
+
Script = c("run_rspec", "run_all_rspec", "run_slow_rspec", "run_old_rspec", "run_rspec_subset"),
|
|
208
|
+
Role = c(
|
|
209
|
+
"Top-level specs/*_spec.rb with spec_helper (see docs/testing.md).",
|
|
210
|
+
"Compile ext/new_bridge; run specs/ and new_bridge_specs/ together.",
|
|
211
|
+
"Suites under slow-specs/ (read script header for spec_helper).",
|
|
212
|
+
"Legacy suites under old_specs/.",
|
|
213
|
+
"Numbered subset 1–18 (Documentation/Spec_Subsets.md)."
|
|
214
|
+
),
|
|
215
|
+
X2 = c("Yes", "Yes", "Yes", "Yes", "Yes"),
|
|
216
|
+
stringsAsFactors = FALSE
|
|
217
|
+
)
|
|
218
|
+
bin_tbl(df_ts)
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
## Other
|
|
222
|
+
|
|
223
|
+
```{r bin-tables-other, echo=FALSE}
|
|
224
|
+
df_ot <- data.frame(
|
|
225
|
+
Script = c("grun", "gstudio_irb.rb / gstudio_pry.rb"),
|
|
226
|
+
Role = c(
|
|
227
|
+
"Graal-era launcher: polyglot ruby with --jvm. Use galaaz-jruby -S instead.",
|
|
228
|
+
"Loaded by gstudio; not meant to be run standalone."
|
|
229
|
+
),
|
|
230
|
+
X2 = c("No", "Yes"),
|
|
231
|
+
stringsAsFactors = FALSE
|
|
232
|
+
)
|
|
233
|
+
bin_tbl(df_ot)
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
For day-to-day **2.0** use, rely on **`bin/galaaz-jruby`**, **`bin/gstudio`**, **`bin/gknit`**, **`bin/run_example`**, **`bin/run_rspec`** / **`bin/run_all_rspec`**, and **`bin/galaaz-bootstrap`** on WSL when using Dockerized R. Treat **`grun`**, **`gknit_Rscript`**, and the polyglot **`ruby`** invocation in **`gknit-draft`** as **legacy** until they are ported to the same JRuby path as **`gknit`**.
|
|
35
237
|
|
|
36
238
|
# System Compatibility
|
|
37
239
|
|
|
38
|
-
|
|
39
|
-
* Ubuntu 18.04 LTS
|
|
40
|
-
* Ubuntu 16.04 LTS
|
|
41
|
-
* Fedora 28
|
|
42
|
-
* macOS 10.14 (Mojave)
|
|
43
|
-
* macOS 10.13 (High Sierra)
|
|
240
|
+
Typical development and CI targets:
|
|
44
241
|
|
|
45
|
-
|
|
242
|
+
* **Linux** — recent Ubuntu LTS or comparable distributions (x86_64).
|
|
243
|
+
* **macOS** — recent releases with JRuby and GNU R available.
|
|
244
|
+
* **Windows** — use **WSL2** (same Linux stack as above); native Windows is not the primary target.
|
|
245
|
+
|
|
246
|
+
The native **gatekeeper** component under `ext/new_bridge` is built with `make` and a C++ toolchain; see the project `README` if compilation fails on your platform.
|
|
46
247
|
|
|
47
|
-
|
|
48
|
-
* FastR
|
|
248
|
+
# Dependencies
|
|
49
249
|
|
|
250
|
+
* **JRuby** — Galaaz 2.0 requires JRuby (tested with **10.1.1.0**) and a matching **JDK** (tested with **Java 21**). MRI Ruby is not supported.
|
|
251
|
+
* **GNU R** — `R` and `Rscript` on your `PATH` (tested with **4.3.3**), plus a C++ toolchain (`g++`, `make`) and the **Rcpp** package to compile the gatekeeper.
|
|
252
|
+
* **galaaz gem** — runtime dependency `msgpack` is pulled in by `gem install`.
|
|
253
|
+
* Optional: **Docker** — if you run R in a container (common on WSL2); see bootstrap below.
|
|
254
|
+
* Optional R packages for examples in this manual — e.g. `ggplot2`, `dplyr`, `knitr`, `kableExtra`, `arrow`, Bioconductor tools such as **DESeq2** (installed the usual R way).
|
|
50
255
|
|
|
51
256
|
# Installation
|
|
52
257
|
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
258
|
+
The supported install is **`gem install` + compile the gatekeeper**. You do not need a git clone.
|
|
259
|
+
|
|
260
|
+
1. Install **JRuby**, a compatible **JDK**, and **GNU R** (with `Rscript` and a C++ compiler).
|
|
261
|
+
2. In R, install **Rcpp**: `install.packages("Rcpp")`.
|
|
262
|
+
3. Install the gem: `jruby -S gem install galaaz`
|
|
263
|
+
4. Compile the native gatekeeper from the installed gem:
|
|
264
|
+
|
|
265
|
+
```
|
|
266
|
+
gem_dir="$(jruby -e "puts Gem::Specification.find_by_name('galaaz').full_gem_path")"
|
|
267
|
+
make -C "${gem_dir}/ext/new_bridge" all
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
5. Ensure **`R`** starts GNU R and can install packages (network access to CRAN when you first call `R.install_and_loads`). For **Apache Arrow** on Java 9+, pass `-J--add-opens=java.base/java.nio=ALL-UNNAMED` to JRuby (from a checkout, `bin/galaaz-jruby` does this).
|
|
271
|
+
|
|
272
|
+
For **gKnit**, **knitr**, **rmarkdown**, and LaTeX (PDF output), install the corresponding R packages, **Pandoc**, and a TeX distribution if you need PDF; the repository includes helpers such as **`bin/install-tinytex`** where appropriate.
|
|
273
|
+
|
|
274
|
+
A **table of all `bin/` scripts** (bootstrap, JRuby wrapper, gstudio, gknit, test runners, and which ones are legacy) is in the section **Command-line tools (`bin/`)** earlier in this manual.
|
|
275
|
+
|
|
276
|
+
### From a repository checkout (contributors)
|
|
277
|
+
|
|
278
|
+
1. Install **bundler** if needed, then run **`jruby -S bundle install`** in the repository root.
|
|
279
|
+
2. Build the bridge native code: **`make -C ext/new_bridge all`** (or **`rake compile_gatekeeper`**).
|
|
280
|
+
3. Run scripts with **`bin/galaaz-jruby`** (sources **`bin/galaaz_jruby_env.inc.sh`** and adds **`-I lib`**).
|
|
281
|
+
|
|
282
|
+
Maintainers can prove a built `.gem` on a throwaway Ubuntu machine (no repo inside the container) with **`./docker/cold-install/run.sh`**.
|
|
283
|
+
|
|
284
|
+
## Windows + WSL2 (optional: Docker / R in a container)
|
|
285
|
+
|
|
286
|
+
If you run Galaaz on Windows through WSL2 and want containerized R instances,
|
|
287
|
+
Docker Desktop is the supported setup.
|
|
288
|
+
|
|
289
|
+
1. Install Docker Desktop on Windows:
|
|
290
|
+
- https://www.docker.com/products/docker-desktop/
|
|
291
|
+
2. Open Docker Desktop and enable WSL integration:
|
|
292
|
+
- Settings > Resources > WSL Integration
|
|
293
|
+
- Enable integration for your target distro
|
|
294
|
+
- Apply & Restart Docker Desktop
|
|
295
|
+
3. In WSL, run Galaaz bootstrap:
|
|
296
|
+
|
|
297
|
+
> ruby bin/galaaz-bootstrap --apply
|
|
298
|
+
> ruby bin/galaaz-bootstrap --check
|
|
299
|
+
|
|
300
|
+
Expected result:
|
|
301
|
+
- docker CLI available
|
|
302
|
+
- docker compose available
|
|
303
|
+
- docker daemon reachable (`docker info` works)
|
|
304
|
+
|
|
305
|
+
If bootstrap reports daemon is unreachable, check Docker Desktop is running and
|
|
306
|
+
WSL integration is enabled for the distro where Galaaz is installed.
|
|
57
307
|
|
|
58
308
|
# Usage
|
|
59
309
|
|
|
@@ -82,8 +332,8 @@ Panda, SciPy, SciKit-Learn and a couple more.
|
|
|
82
332
|
|
|
83
333
|
> galaaz -T
|
|
84
334
|
|
|
85
|
-
Shows a list with all available
|
|
86
|
-
|
|
335
|
+
Shows a list with all available executable tasks. To execute a task, substitute the
|
|
336
|
+
'rake' word in the list with 'galaaz'. For instance, the following line shows up
|
|
87
337
|
after 'galaaz -T'
|
|
88
338
|
|
|
89
339
|
rake master_list:scatter_plot # scatter_plot from:....
|
|
@@ -92,18 +342,948 @@ Panda, SciPy, SciKit-Learn and a couple more.
|
|
|
92
342
|
|
|
93
343
|
> galaaz master_list:scatter_plot
|
|
94
344
|
|
|
345
|
+
# JRuby, multithreading, and the R bridge
|
|
346
|
+
|
|
347
|
+
Galaaz 2.0 runs Ruby on **JRuby**, so your application can use **real parallel threads** for
|
|
348
|
+
I/O-bound work (HTTP clients, database connections, message consumers, and so on). R itself is
|
|
349
|
+
still executed in a **single GNU R process** behind the Galaaz bridge.
|
|
350
|
+
|
|
351
|
+
When several Ruby threads call into R at the same time, the bridge **serializes** those calls:
|
|
352
|
+
each request is matched to a reply using an internal per-call **queue**, so you do not need to
|
|
353
|
+
add your own mutex around every `R.foo` from application threads. (You should still use normal
|
|
354
|
+
Ruby synchronization when **Ruby** data structures are shared between threads—for example, when
|
|
355
|
+
appending rows from each thread into a shared array before sending them to R.)
|
|
356
|
+
|
|
357
|
+
A practical pattern is:
|
|
358
|
+
|
|
359
|
+
1. Use threads (or a connection pool) to read from **multiple databases or shards** in parallel.
|
|
360
|
+
2. Merge the rows in Ruby under a `Mutex` if you collect into one structure.
|
|
361
|
+
3. Hand the merged table to R **once** (for example with `R::Arrow.from_ruby_batches` and dplyr,
|
|
362
|
+
or by building a data frame) so heavy statistics run in R with fewer bridge round-trips.
|
|
363
|
+
|
|
364
|
+
A runnable sketch lives in
|
|
365
|
+
`examples/multithread_shards_to_r/shards_to_r.rb` (simulated shard queries; swap in your DB
|
|
366
|
+
driver). For concurrency tests on the bridge itself, see `specs/bridge_concurrent_spec.rb` and
|
|
367
|
+
`specs/arrow_from_ruby_batches_spec.rb`.
|
|
368
|
+
|
|
369
|
+
## Long-running R calls and a completion block
|
|
370
|
+
|
|
371
|
+
For R work that can take a long time, the bridge can avoid a Ruby-side **wait timeout** by
|
|
372
|
+
scheduling the call and resuming in a **block** when the `RET` arrives.
|
|
373
|
+
|
|
374
|
+
- **`R.eval_r_async(code, timeout: nil) { |result| ... }`** — string eval; on success, `result.value`
|
|
375
|
+
is the same formatted string as **`R.eval_r`** (use `timeout: nil` for no Ruby-side limit).
|
|
376
|
+
- **`R::Async.<rname>(...) { |result| ... }`** — same dispatch as **`R.<rname>(...)`**, but async;
|
|
377
|
+
on success, `result.value` is an **`R::Object`** (or unboxed Ruby value / Symbol), like synchronous
|
|
378
|
+
**`R.<rname>`**. Optional keyword **`timeout:`** applies a Ruby-side wait limit (completion receives
|
|
379
|
+
**`NewBridge::SessionClient::TimeoutError`** if R is too slow).
|
|
380
|
+
|
|
381
|
+
**Important:** **`R.foo(...) { |x| }`** is already used for dplyr-style scopes (`R::Support.new_scope`),
|
|
382
|
+
so async R calls must use **`R::Async`** or **`R.eval_r_async`**, not a bare **`R.foo` with a block.**
|
|
383
|
+
|
|
384
|
+
`NewBridge::EvalResult` exposes **`#ok?`**, **`#value`**, and **`#error`**. The completion block runs on a
|
|
385
|
+
**background thread** (not the bridge reader thread).
|
|
386
|
+
|
|
387
|
+
The example below is **plain Ruby** (no Rails). The R snippet sleeps (standing in for heavy work) and then
|
|
388
|
+
returns an integer so the success branch shows a **non-nil** value. (`Sys.sleep` alone returns **NULL** in R;
|
|
389
|
+
on success **`result.value`** is then **`nil`** in Ruby—that is expected, not a bridge error.)
|
|
390
|
+
|
|
391
|
+
```{ruby long_r_completion_block}
|
|
392
|
+
require 'thread'
|
|
393
|
+
|
|
394
|
+
completion = Queue.new
|
|
395
|
+
|
|
396
|
+
R.eval_r_async('({ Sys.sleep(0.3); 42L })', timeout: nil) do |result|
|
|
397
|
+
if result.ok?
|
|
398
|
+
puts "[completion] R finished; eval_r-style value: #{result.value.inspect}"
|
|
399
|
+
else
|
|
400
|
+
puts "[completion] R/bridge error: #{result.error.class}: #{result.error.message}"
|
|
401
|
+
end
|
|
402
|
+
completion.push(:done)
|
|
403
|
+
end
|
|
404
|
+
|
|
405
|
+
3.times do |i|
|
|
406
|
+
puts "[main] other Ruby work step #{i + 1}"
|
|
407
|
+
sleep 0.05
|
|
408
|
+
end
|
|
409
|
+
|
|
410
|
+
completion.pop
|
|
411
|
+
puts "[main] R completion has run; exiting."
|
|
412
|
+
```
|
|
413
|
+
|
|
414
|
+
In a **web application**, the HTTP response usually ends before R finishes, so you would not
|
|
415
|
+
`Queue#pop` in the controller; you would persist an identifier, let the completion block write
|
|
416
|
+
the outcome to storage, and notify the client (poll, WebSocket, Turbo Stream, etc.). The plain
|
|
417
|
+
Ruby pattern above is only to show **when** the result exists (inside the block, or after data
|
|
418
|
+
written there is observed elsewhere). Runnable specs live in **`new_bridge_specs/eval_r_async_spec.rb`**.
|
|
419
|
+
|
|
420
|
+
## Galaaz + Rails (JRuby) integration baseline
|
|
421
|
+
|
|
422
|
+
This section documents the baseline we used to create a working Rails app with Galaaz in WSL.
|
|
423
|
+
The goals were:
|
|
424
|
+
|
|
425
|
+
1. Rails boots under **JRuby**.
|
|
426
|
+
2. Galaaz is loaded from a local checkout (before publishing to RubyGems).
|
|
427
|
+
3. A request path can execute **`R.eval(...)`** and return a result.
|
|
428
|
+
|
|
429
|
+
### 1) Create the app with JRuby-friendly options
|
|
430
|
+
|
|
431
|
+
Rails defaults can pull gems that are not ideal on JRuby-first setups (for example sqlite native
|
|
432
|
+
extension paths and deployment extras). A minimal app avoids early friction:
|
|
433
|
+
|
|
434
|
+
```bash
|
|
435
|
+
cd /home/rbotafogo/desenv_linux
|
|
436
|
+
jruby -S rails new hedi --skip-git --minimal --skip-kamal --skip-solid --skip-active-record
|
|
437
|
+
```
|
|
438
|
+
|
|
439
|
+
Then install gems:
|
|
440
|
+
|
|
441
|
+
```bash
|
|
442
|
+
cd /home/rbotafogo/desenv_linux/hedi
|
|
443
|
+
jruby -S bundle install
|
|
444
|
+
```
|
|
445
|
+
|
|
446
|
+
### 2) Use Galaaz as a local path gem
|
|
447
|
+
|
|
448
|
+
For local development we keep a stable path:
|
|
449
|
+
|
|
450
|
+
- `~/gems/galaaz` -> symlink to your Galaaz checkout
|
|
451
|
+
- optional built gem archive in `~/gems/pkg/`
|
|
452
|
+
|
|
453
|
+
In Rails `Gemfile`:
|
|
454
|
+
|
|
455
|
+
```ruby
|
|
456
|
+
gem "galaaz", path: "/home/rbotafogo/gems/galaaz", require: false
|
|
457
|
+
```
|
|
458
|
+
|
|
459
|
+
And load after Rails boot in `config/application.rb`:
|
|
460
|
+
|
|
461
|
+
```ruby
|
|
462
|
+
config.after_initialize { require "galaaz" }
|
|
463
|
+
```
|
|
464
|
+
|
|
465
|
+
Why `require: false` + `after_initialize`? In this integration, loading Galaaz too early via
|
|
466
|
+
`Bundler.require` triggered Rails/JRuby initialization failures.
|
|
467
|
+
|
|
468
|
+
### 3) Simple request-path smoke test
|
|
469
|
+
|
|
470
|
+
A direct smoke test from Rails runner:
|
|
471
|
+
|
|
472
|
+
```bash
|
|
473
|
+
cd /home/rbotafogo/desenv_linux/hedi
|
|
474
|
+
jruby -S bundle exec rails runner "puts R.eval('sum(c(1,2,3,4,5))').inspect"
|
|
475
|
+
```
|
|
476
|
+
|
|
477
|
+
Expected output:
|
|
478
|
+
|
|
479
|
+
```text
|
|
480
|
+
15.0
|
|
481
|
+
```
|
|
482
|
+
|
|
483
|
+
### 4) HTTP endpoint pattern
|
|
484
|
+
|
|
485
|
+
For this baseline, a small Rack endpoint was the most stable first step to prove request-time R
|
|
486
|
+
evaluation. (A full ActionController stack can be enabled later as the app evolves.)
|
|
487
|
+
|
|
488
|
+
Minimal pattern:
|
|
489
|
+
|
|
490
|
+
1. Define a Rack app class under `lib/` that runs `R.eval(...)` and returns HTML/JSON.
|
|
491
|
+
2. Point a route to that Rack app (`root to: MyRackApp`).
|
|
492
|
+
3. Verify with browser/curl.
|
|
493
|
+
|
|
494
|
+
### 5) Running from WSL and opening from Windows
|
|
495
|
+
|
|
496
|
+
Recommended bind:
|
|
497
|
+
|
|
498
|
+
```bash
|
|
499
|
+
jruby -S bundle exec rails server -b 0.0.0.0 -p 3000
|
|
500
|
+
```
|
|
501
|
+
|
|
502
|
+
Then open from Windows:
|
|
503
|
+
|
|
504
|
+
- `http://localhost:3000` (usually works with WSL localhost forwarding), or
|
|
505
|
+
- `http://<wsl-ip>:3000` if needed.
|
|
506
|
+
|
|
507
|
+
In development, if Host Authorization blocks requests with unexpected Host headers, use:
|
|
508
|
+
|
|
509
|
+
```ruby
|
|
510
|
+
# config/environments/development.rb
|
|
511
|
+
config.hosts.clear
|
|
512
|
+
```
|
|
513
|
+
|
|
514
|
+
### 6) Troubleshooting checklist
|
|
515
|
+
|
|
516
|
+
- `Could not find ... in locally installed gems`:
|
|
517
|
+
run `jruby -S bundle install` in the Rails app directory.
|
|
518
|
+
- Stale PID after crash:
|
|
519
|
+
remove `tmp/pids/server.pid`.
|
|
520
|
+
- Local Galaaz path changed:
|
|
521
|
+
verify `Gemfile` path target exists and rerun bundler.
|
|
522
|
+
- R runtime issues:
|
|
523
|
+
confirm GNU R is installed and on `PATH` in the same shell where Rails runs.
|
|
524
|
+
|
|
525
|
+
As new Rails features are added (controllers, jobs, websockets, background rendering, plot
|
|
526
|
+
generation), extend this section with concrete, runnable snippets and the associated operational
|
|
527
|
+
checks.
|
|
528
|
+
|
|
529
|
+
# Accessing R from Ruby
|
|
530
|
+
|
|
531
|
+
One of the nice aspects of Galaaz is that variables and functions defined in R can
|
|
532
|
+
be easily accessed from Ruby. For instance, to access the `mtcars` data frame from R
|
|
533
|
+
in Ruby, we use the symbol `:mtcars` preceded by the `~` operator: `~R[:mtcars]` retrieves the
|
|
534
|
+
value of the `mtcars` object in R.
|
|
535
|
+
|
|
536
|
+
```{ruby access_r}
|
|
537
|
+
puts ~R[:mtcars]
|
|
538
|
+
```
|
|
539
|
+
|
|
540
|
+
## Scoped symbols and lexical scoping
|
|
541
|
+
|
|
542
|
+
Galaaz 2.0 uses **scoped symbols** by default. The canonical style is `R[:name]`:
|
|
543
|
+
|
|
544
|
+
- `~R[:mtcars]` fetches an R object by name.
|
|
545
|
+
- `R[:a] + R[:b]` builds an expression.
|
|
546
|
+
- `R[:year].up_to(R[:day])` builds range expressions.
|
|
547
|
+
|
|
548
|
+
If you prefer the terse `:x` syntax, you can opt in with lexical scoping using a Ruby refinement:
|
|
549
|
+
|
|
550
|
+
```ruby
|
|
551
|
+
module MyScript
|
|
552
|
+
using Galaaz::SymbolDSL
|
|
553
|
+
|
|
554
|
+
def self.run
|
|
555
|
+
expr = :a + :b
|
|
556
|
+
puts expr
|
|
557
|
+
puts ~:mtcars
|
|
558
|
+
end
|
|
559
|
+
end
|
|
560
|
+
```
|
|
561
|
+
|
|
562
|
+
`using Galaaz::SymbolDSL` is **lexically scoped**: only code in that module/file scope gets `:x` DSL behavior.
|
|
563
|
+
Outside that scope, plain Ruby `Symbol` behavior is unchanged.
|
|
564
|
+
|
|
565
|
+
To access an R function from Ruby, the R function needs to be preceded by `R.` scoping.
|
|
566
|
+
Below we see an example of creating a R::Vector by calling the 'c' R function
|
|
567
|
+
|
|
568
|
+
```{ruby call_r_func}
|
|
569
|
+
puts vec = R.c(1.0, 2.0, 3.0, 4.0)
|
|
570
|
+
```
|
|
571
|
+
Note that 'vec' is an object of type R::Vector:
|
|
572
|
+
|
|
573
|
+
```{ruby r_object}
|
|
574
|
+
puts vec.class
|
|
575
|
+
```
|
|
576
|
+
Every object created by a call to an R function will be of a type that inherits from
|
|
577
|
+
R::Object. In R, there is also a function 'class'. In order to access that function we
|
|
578
|
+
can call method 'rclass' in the R::Object:
|
|
579
|
+
|
|
580
|
+
```{ruby rclass}
|
|
581
|
+
puts vec.rclass
|
|
582
|
+
```
|
|
583
|
+
When working with R::Object(s), it is possible to use the '.' operator to pipe operations.
|
|
584
|
+
When using '.', the object to which the '.' is applied becomes the first argument of the
|
|
585
|
+
corresponding R function. For instance, function 'c' in R, can be used to concatenate
|
|
586
|
+
two vectors or more vectors (in R, there are no scalar values, scalars are converted to
|
|
587
|
+
vectors of size 1. Within Galaaz, scalar parameter is converted to a size one vector):
|
|
588
|
+
|
|
589
|
+
```{ruby concat}
|
|
590
|
+
puts R.c(vec, 10, 20, 30)
|
|
591
|
+
```
|
|
592
|
+
The call above to the 'c' function can also be done using '.' notation:
|
|
593
|
+
|
|
594
|
+
```{ruby concat_with_dot}
|
|
595
|
+
puts vec.c(10, 20, 30)
|
|
596
|
+
```
|
|
597
|
+
We will talk about vector indexing in a later section. But notice here that indexing
|
|
598
|
+
an R::Vector will return another R::Vector:
|
|
599
|
+
|
|
600
|
+
```{ruby indexing}
|
|
601
|
+
puts vec[1]
|
|
602
|
+
```
|
|
603
|
+
Sometimes we want to index an R::Object and get back a Ruby object that is not wrapped
|
|
604
|
+
in an R::Object, but the native Ruby object. For this, we can index the R object with
|
|
605
|
+
the '>>' operator:
|
|
606
|
+
|
|
607
|
+
```{ruby native_value}
|
|
608
|
+
puts vec >> 0
|
|
609
|
+
puts vec >> 2
|
|
610
|
+
```
|
|
611
|
+
|
|
612
|
+
It is also possible to call an R function with named arguments, by creating the function
|
|
613
|
+
in Galaaz with named parameters. For instance, here is an example of creating a 'list'
|
|
614
|
+
with named elements:
|
|
615
|
+
|
|
616
|
+
```{ruby named_parameters}
|
|
617
|
+
puts R.list(first_name: "Rodrigo", last_name: "Botafogo")
|
|
618
|
+
```
|
|
619
|
+
|
|
620
|
+
Many R functions receive another function as argument. For instance, method 'map' applies
|
|
621
|
+
a function to every element of a vector. With Galaaz, it is possible to pass a Proc,
|
|
622
|
+
Method or Lambda in place of the expected R function. In this next example, we will
|
|
623
|
+
add 2 to every element of our previously created vector:
|
|
624
|
+
|
|
625
|
+
```{ruby proc_as_param}
|
|
626
|
+
puts vec.map { |x| x + 2 }
|
|
627
|
+
```
|
|
628
|
+
|
|
95
629
|
# gKnitting a Document
|
|
96
630
|
|
|
97
|
-
This manual has been formatted
|
|
98
|
-
a document in Ruby or R and output it in any of the available formats for R
|
|
99
|
-
gKnit runs
|
|
631
|
+
This manual has been formatted using gKnit. gKnit uses knitr and R Markdown to knit
|
|
632
|
+
a document in Ruby or R and output it in any of the available formats for R Markdown.
|
|
633
|
+
gKnit runs with **JRuby**, **GNU R**, and Galaaz. In gKnit, Ruby variables are persisted between
|
|
100
634
|
chunks, making it an ideal solution for literate programming. Also, since it is based
|
|
101
|
-
on Galaaz, Ruby chunks can have access to R variables and
|
|
102
|
-
|
|
635
|
+
on Galaaz, Ruby chunks can have access to R variables and combining Ruby with R in one
|
|
636
|
+
document is natural.
|
|
637
|
+
|
|
638
|
+
The idea of "literate programming" was first introduced by Donald Knuth in the
|
|
639
|
+
1980's [@Knuth:literate_programming].
|
|
640
|
+
The main intention of this approach was to develop software interspersing macro snippets,
|
|
641
|
+
traditional source code, and a natural language such as English in a document
|
|
642
|
+
that could be compiled into
|
|
643
|
+
executable code and at the same time easily read by a human developer. According to Knuth
|
|
644
|
+
"The practitioner of
|
|
645
|
+
literate programming can be regarded as an essayist, whose main concern is with exposition
|
|
646
|
+
and excellence of style."
|
|
647
|
+
|
|
648
|
+
The idea of literate programming evolved into the idea of reproducible research, in which
|
|
649
|
+
all the data, software code, documentation, graphics etc. needed to reproduce the research
|
|
650
|
+
and its reports could be included in a
|
|
651
|
+
single document or set of documents that when distributed to peers could be rerun generating
|
|
652
|
+
the same output and reports.
|
|
653
|
+
|
|
654
|
+
The R community has put a great deal of effort in reproducible research. In 2002, Sweave was
|
|
655
|
+
introduced and it allowed mixing R code with LaTeX, generating high-quality PDF documents. A
|
|
656
|
+
Sweave document could include code, the results of executing the code, graphics and text
|
|
657
|
+
such that it contained the whole narrative to reproduce the research. In
|
|
658
|
+
2012, Knitr, developed by Yihui Xie from RStudio was released to replace Sweave and to
|
|
659
|
+
consolidate in one single package the many extensions and add-on packages that
|
|
660
|
+
were necessary for Sweave.
|
|
661
|
+
|
|
662
|
+
With Knitr, __R markdown__ was also developed, an extension to the
|
|
663
|
+
Markdown format. With __R markdown__ and Knitr it is possible to generate reports in a multitude
|
|
664
|
+
of formats such as HTML, Markdown, LaTeX, PDF, DVI, etc. __R markdown__ also allows the use of
|
|
665
|
+
multiple programming languages such as R, Ruby, Python, etc. in the same document.
|
|
666
|
+
|
|
667
|
+
In __R markdown__, text is interspersed with
|
|
668
|
+
code chunks that can be executed and both the code and its results can become
|
|
669
|
+
part of the final report. Although __R markdown__ allows multiple programming languages in the
|
|
670
|
+
same document, only R and Python (with
|
|
671
|
+
the reticulate package) can persist variables between chunks. For other languages, such as
|
|
672
|
+
Ruby, every chunk will start a new process and thus all data is lost between chunks, unless it
|
|
673
|
+
is somehow stored in a data file that is read by the next chunk.
|
|
674
|
+
|
|
675
|
+
Being able to persist data
|
|
676
|
+
between chunks is critical for literate programming otherwise the flow of the narrative is lost
|
|
677
|
+
by all the effort of having to save data and then reload it. Although this might, at first, seem like
|
|
678
|
+
a small nuisance, not being able to persist data between chunks is a major issue. For example, let's
|
|
679
|
+
take a look at the following simple example in which we want to show how to create a list and the
|
|
680
|
+
use it. Let's first assume that data cannot be persisted between chunks. In the next chunk we
|
|
681
|
+
create a list, then we would need to save it to file, but to save it, we need somehow to marshal the
|
|
682
|
+
data into a binary format:
|
|
683
|
+
|
|
684
|
+
```{ruby no_persistence}
|
|
685
|
+
lst = R.list(a: 1, b: 2, c: 3)
|
|
686
|
+
lst.saveRDS("lst.rds")
|
|
687
|
+
```
|
|
688
|
+
then, on the next chunk, where variable 'lst' is used, we need to read back it's value
|
|
689
|
+
|
|
690
|
+
```{ruby load_persisted_data}
|
|
691
|
+
lst = R.readRDS("lst.rds")
|
|
692
|
+
puts lst
|
|
693
|
+
```
|
|
694
|
+
|
|
695
|
+
Now, any single code has dozens of variables that we might want to use and reuse between chunks.
|
|
696
|
+
Clearly, such an approach becomes quickly unmanageable. Probably, because of
|
|
697
|
+
this problem, it is very rare to see any __R markdown__ document in the Ruby community.
|
|
698
|
+
|
|
699
|
+
When variables can be used across chunks, then no overhead is needed:
|
|
700
|
+
|
|
701
|
+
```{ruby persistence}
|
|
702
|
+
lst = R.list(a: 1, b: 2, c: 3)
|
|
703
|
+
# any other code can be added here
|
|
704
|
+
```
|
|
705
|
+
|
|
706
|
+
```{ruby use_var}
|
|
707
|
+
puts lst
|
|
708
|
+
```
|
|
709
|
+
|
|
710
|
+
In the Python community, the same effort to have code and text in an integrated environment
|
|
711
|
+
started around the first decade of the 2000s. In 2006 IPython 0.7.2 was released. In 2014,
|
|
712
|
+
Fernando Pérez spun off the Jupyter project from IPython, creating a web-based interactive
|
|
713
|
+
computation environment. Jupyter can now be used with many languages, including Ruby with the
|
|
714
|
+
iruby gem (https://github.com/SciRuby/iruby). In order to have multiple languages in a Jupyter
|
|
715
|
+
notebook the SoS kernel was developed (https://vatlab.github.io/sos-docs/).
|
|
716
|
+
|
|
717
|
+
## gKnit and __R markdown__
|
|
718
|
+
|
|
719
|
+
gKnit is based on knitr and __R markdown__ and can knit a document
|
|
720
|
+
written both in Ruby and/or R and output it in any of the available formats of __R markdown__. gKnit
|
|
721
|
+
allows ruby developers to do literate programming and reproducible research by allowing them to
|
|
722
|
+
have in a single document, text and code.
|
|
723
|
+
|
|
724
|
+
In gKnit, Ruby variables are persisted between
|
|
725
|
+
chunks, making it an ideal solution for literate programming in this language. Also,
|
|
726
|
+
since it is based on Galaaz, Ruby chunks can access R variables (`~R[:name]`, `R.*`) through the
|
|
727
|
+
**Galaaz bridge** while knitr drives **GNU R**—no GraalVM polyglot runtime is required.
|
|
728
|
+
|
|
729
|
+
This is not a blog post on __R markdown__, and the interested user is directed to the following links
|
|
730
|
+
for detailed information on its capabilities and use.
|
|
731
|
+
|
|
732
|
+
* https://rmarkdown.rstudio.com/ or
|
|
733
|
+
* https://bookdown.org/yihui/rmarkdown/
|
|
734
|
+
|
|
735
|
+
In this post, we will describe just the main aspects of __R markdown__, so the user can start
|
|
736
|
+
gKnitting Ruby and R documents quickly.
|
|
737
|
+
|
|
738
|
+
## The Yaml header
|
|
739
|
+
|
|
740
|
+
An __R markdown__ document should start with a Yaml header and be stored in a file with
|
|
741
|
+
'.Rmd' extension. This document has the following header for gKnitting an HTML document.
|
|
742
|
+
|
|
743
|
+
```
|
|
744
|
+
---
|
|
745
|
+
title: "How to do reproducible research in Ruby with gKnit"
|
|
746
|
+
author:
|
|
747
|
+
- "Rodrigo Botafogo"
|
|
748
|
+
- "Daniel Mossé - University of Pittsburgh"
|
|
749
|
+
tags: [Tech, Data Science, Ruby, R, JRuby, Galaaz]
|
|
750
|
+
date: "20/02/2019"
|
|
751
|
+
output:
|
|
752
|
+
html_document:
|
|
753
|
+
self_contained: true
|
|
754
|
+
keep_md: true
|
|
755
|
+
pdf_document:
|
|
756
|
+
includes:
|
|
757
|
+
in_header: ["../../sty/galaaz.sty"]
|
|
758
|
+
number_sections: yes
|
|
759
|
+
---
|
|
760
|
+
```
|
|
761
|
+
|
|
762
|
+
For more information on the options in the Yaml header, [check here](https://bookdown.org/yihui/rmarkdown/html-document.html).
|
|
763
|
+
|
|
764
|
+
## Choosing the output format when calling gknit
|
|
765
|
+
|
|
766
|
+
Yes: you can select the render target on the **command line**. **`bin/gknit`** (or **`gknit`** on your `PATH`) forwards options to **`rmarkdown::render`** via **`R::Rmarkdown.render`**.
|
|
767
|
+
|
|
768
|
+
* **`--output_format FORMAT`** — name of the format, as in the YAML `output:` block. Examples:
|
|
769
|
+
* **`html_document`** — HTML (often the default you list first under `output:`).
|
|
770
|
+
* **`pdf_document`** — PDF (you need a working LaTeX setup, e.g. TinyTeX; see **`bin/install-tinytex`**).
|
|
771
|
+
* **`md_document`**, **`github_document`**, or any other format defined in your YAML.
|
|
772
|
+
* **`all`** — render **every** format declared under `output:` in the document (same idea as in R Markdown).
|
|
773
|
+
|
|
774
|
+
If you **omit** **`--output_format`**, gknit passes **`NULL`** for the format argument. In that case **rmarkdown** uses the **first** format listed under **`output:`** in the YAML (and if none is specified there, behavior follows the usual rmarkdown defaults, typically HTML).
|
|
775
|
+
|
|
776
|
+
Other useful flags:
|
|
777
|
+
|
|
778
|
+
* **`--output_file NAME`** — output file name (optional path; see also **`--output_dir`**).
|
|
779
|
+
* **`--output_dir DIR`** — directory for the rendered file (created if missing).
|
|
780
|
+
* **`--bridge_timeout_sec`** / **`--callback_timeout_ms`** — longer R or install steps (see elsewhere in this manual).
|
|
781
|
+
|
|
782
|
+
Examples (run from the directory where paths make sense, or use absolute paths):
|
|
783
|
+
|
|
784
|
+
```text
|
|
785
|
+
bin/gknit blogs/manual/manual.Rmd
|
|
786
|
+
bin/gknit --output_format html_document blogs/manual/manual.Rmd
|
|
787
|
+
bin/gknit --output_format pdf_document blogs/manual/manual.Rmd
|
|
788
|
+
bin/gknit --output_format all blogs/manual/manual.Rmd
|
|
789
|
+
```
|
|
790
|
+
|
|
791
|
+
Use **`gknit -h`** for the full option list.
|
|
792
|
+
|
|
793
|
+
## __R Markdown__ formatting
|
|
794
|
+
|
|
795
|
+
Document formatting can be done with simple markups such as:
|
|
796
|
+
|
|
797
|
+
## Headers
|
|
798
|
+
|
|
799
|
+
```
|
|
800
|
+
# Header 1
|
|
103
801
|
|
|
104
|
-
|
|
802
|
+
## Header 2
|
|
105
803
|
|
|
106
|
-
|
|
804
|
+
### Header 3
|
|
805
|
+
|
|
806
|
+
```
|
|
807
|
+
|
|
808
|
+
## Lists
|
|
809
|
+
|
|
810
|
+
```
|
|
811
|
+
Unordered lists:
|
|
812
|
+
|
|
813
|
+
* Item 1
|
|
814
|
+
* Item 2
|
|
815
|
+
+ Item 2a
|
|
816
|
+
+ Item 2b
|
|
817
|
+
```
|
|
818
|
+
|
|
819
|
+
```
|
|
820
|
+
Ordered Lists
|
|
821
|
+
|
|
822
|
+
1. Item 1
|
|
823
|
+
2. Item 2
|
|
824
|
+
3. Item 3
|
|
825
|
+
+ Item 3a
|
|
826
|
+
+ Item 3b
|
|
827
|
+
```
|
|
828
|
+
|
|
829
|
+
For more R markdown formatting go to https://rmarkdown.rstudio.com/authoring_basics.html.
|
|
830
|
+
|
|
831
|
+
## R chunks
|
|
832
|
+
|
|
833
|
+
Running and executing Ruby and R code is actually what really interests us is this blog.
|
|
834
|
+
Inserting a code chunk is done by adding code in a block delimited by three back ticks
|
|
835
|
+
followed by an open
|
|
836
|
+
curly brace ('{') followed with the engine name (r, ruby, rb, include, ...), an
|
|
837
|
+
any optional chunk_label and options, as shown below:
|
|
838
|
+
|
|
839
|
+
````
|
|
840
|
+
```{engine_name [chunk_label], [chunk_options]}`r ''`
|
|
841
|
+
```
|
|
842
|
+
````
|
|
843
|
+
|
|
844
|
+
for instance, let's add an R chunk to the document labeled 'first_r_chunk'. This is
|
|
845
|
+
a very simple code just to create a variable and print it out, as follows:
|
|
846
|
+
|
|
847
|
+
````
|
|
848
|
+
```{r first_r_chunk}`r ''`
|
|
849
|
+
vec <- c(1, 2, 3)
|
|
850
|
+
print(vec)
|
|
851
|
+
```
|
|
852
|
+
````
|
|
853
|
+
|
|
854
|
+
If this block is added to an __R markdown__ document and gKnitted the result will be:
|
|
855
|
+
|
|
856
|
+
```{r first_r_chunk}
|
|
857
|
+
vec <- c(1, 2, 3)
|
|
858
|
+
print(vec)
|
|
859
|
+
```
|
|
860
|
+
|
|
861
|
+
Now let's say that we want to do some analysis in the code, but just print the result and not the
|
|
862
|
+
code itself. For this, we need to add the option 'echo = FALSE'.
|
|
863
|
+
|
|
864
|
+
````
|
|
865
|
+
```{r second_r_chunk, echo = FALSE}`r ''`
|
|
866
|
+
vec2 <- c(10, 20, 30)
|
|
867
|
+
vec3 <- vec * vec2
|
|
868
|
+
print(vec3)
|
|
869
|
+
```
|
|
870
|
+
````
|
|
871
|
+
Here is how this block will show up in the document. Observe that the code is not shown
|
|
872
|
+
and we only see the execution result in a white box
|
|
873
|
+
|
|
874
|
+
```{r second_r_chunk, echo = FALSE}
|
|
875
|
+
vec2 <- c(10, 20, 30)
|
|
876
|
+
vec3 <- vec * vec2
|
|
877
|
+
print(vec3)
|
|
878
|
+
```
|
|
879
|
+
|
|
880
|
+
A description of the available chunk options can be found in https://yihui.name/knitr/.
|
|
881
|
+
|
|
882
|
+
Let's add another R chunk with a function definition. In this example, a vector
|
|
883
|
+
'r_vec' is created and
|
|
884
|
+
a new function 'reduce_sum' is defined. The chunk specification is
|
|
885
|
+
|
|
886
|
+
````
|
|
887
|
+
```{r data_creation}`r ''`
|
|
888
|
+
r_vec <- c(1, 2, 3, 4, 5)
|
|
889
|
+
|
|
890
|
+
reduce_sum <- function(...) {
|
|
891
|
+
Reduce(sum, as.list(...))
|
|
892
|
+
}
|
|
893
|
+
```
|
|
894
|
+
````
|
|
895
|
+
|
|
896
|
+
and this is how it will look like once executed. From now on, to be concise in the
|
|
897
|
+
presentation we will not show chunk definitions any longer.
|
|
898
|
+
|
|
899
|
+
|
|
900
|
+
```{r data_creation}
|
|
901
|
+
r_vec <- c(1, 2, 3, 4, 5)
|
|
902
|
+
|
|
903
|
+
reduce_sum <- function(...) {
|
|
904
|
+
Reduce(sum, as.list(...))
|
|
905
|
+
}
|
|
906
|
+
```
|
|
907
|
+
|
|
908
|
+
We can, possibly in another chunk, access the vector and call the function as follows:
|
|
909
|
+
|
|
910
|
+
```{r using_previous}
|
|
911
|
+
print(r_vec)
|
|
912
|
+
print(reduce_sum(r_vec))
|
|
913
|
+
```
|
|
914
|
+
## R Graphics with ggplot
|
|
915
|
+
|
|
916
|
+
In the following chunk, we create a bubble chart in R using ggplot and include it in
|
|
917
|
+
this document. Note that there is no directive in the code to include the image, this
|
|
918
|
+
occurs automatically. The 'mpg' dataframe is natively available to R and to Galaaz as
|
|
919
|
+
well.
|
|
920
|
+
|
|
921
|
+
For the reader not knowledgeable of ggplot, ggplot is a graphics library based on "the
|
|
922
|
+
grammar of graphics" [@Wilkinson:grammar_of_graphics]. The idea of the grammar of graphics
|
|
923
|
+
is to build a graphics by adding layers to the plot. More information can be found in
|
|
924
|
+
https://towardsdatascience.com/a-comprehensive-guide-to-the-grammar-of-graphics-for-effective-visualization-of-multi-dimensional-1f92b4ed4149.
|
|
925
|
+
|
|
926
|
+
In the plot below the 'mpg' dataset from base R is used. "The data concerns city-cycle fuel
|
|
927
|
+
consumption in miles per gallon, to be predicted in terms of 3 multivalued discrete and 5
|
|
928
|
+
continuous attributes." (Quinlan, 1993)
|
|
929
|
+
|
|
930
|
+
First, the 'mpg' dataset if filtered to extract only cars from the following manumactures: Audi, Ford,
|
|
931
|
+
Honda, and Hyundai and stored in the 'mpg_select' variable. Then, the selected dataframe is passed
|
|
932
|
+
to the ggplot function specifying in the aesthetic method (aes) that 'displacement' (disp) should
|
|
933
|
+
be plotted in the 'x' axis and 'city mileage' should be on the 'y' axis. In the 'labs' layer we
|
|
934
|
+
pass the 'title' and 'subtitle' for the plot. To the basic plot 'g', geom\_jitter is added, that
|
|
935
|
+
plots cars from the same manufactures with the same color (col=manufactures) and the size of the
|
|
936
|
+
car point equal its high way consumption (size = hwy). Finally, a last layer is plotter containing
|
|
937
|
+
a linear regression line (method = "lm") for every manufacturer.
|
|
938
|
+
|
|
939
|
+
```{r bubble, dev='png'}
|
|
940
|
+
# load package and data
|
|
941
|
+
library(ggplot2)
|
|
942
|
+
data(mpg, package="ggplot2")
|
|
943
|
+
|
|
944
|
+
mpg_select <- mpg[mpg$manufacturer %in% c("audi", "ford", "honda", "hyundai"), ]
|
|
945
|
+
|
|
946
|
+
# Scatterplot
|
|
947
|
+
theme_set(theme_bw()) # pre-set the bw theme.
|
|
948
|
+
g <- ggplot(mpg_select, aes(displ, cty)) +
|
|
949
|
+
labs(subtitle="mpg: Displacement vs City Mileage",
|
|
950
|
+
title="Bubble chart")
|
|
951
|
+
|
|
952
|
+
g + geom_jitter(aes(col=manufacturer, size=hwy)) +
|
|
953
|
+
geom_smooth(aes(col=manufacturer), method="lm", se=F)
|
|
954
|
+
```
|
|
955
|
+
|
|
956
|
+
## Ruby chunks
|
|
957
|
+
|
|
958
|
+
Including a Ruby chunk is just as easy as including an R chunk in the document: just
|
|
959
|
+
change the name of the engine to 'ruby'. It is also possible to pass chunk options
|
|
960
|
+
to the Ruby engine; however, this version does not accept all the options that are
|
|
961
|
+
available to R chunks. Future versions will add those options.
|
|
962
|
+
|
|
963
|
+
````
|
|
964
|
+
```{ruby first_ruby_chunk}`r ''`
|
|
965
|
+
```
|
|
966
|
+
````
|
|
967
|
+
|
|
968
|
+
In this example, the ruby chunk is called 'first_ruby_chunk'. One important
|
|
969
|
+
aspect of chunk labels is that they cannot be duplicated. If a chunk label is
|
|
970
|
+
duplicated, gKnit will stop with an error.
|
|
971
|
+
|
|
972
|
+
In the following chunk, variable 'a', 'b' and 'c' are standard Ruby variables
|
|
973
|
+
and 'vec' and 'vec2' are two vectors created by calling the 'c' method on the
|
|
974
|
+
R module.
|
|
975
|
+
|
|
976
|
+
In Galaaz, the R module allows us to access R functions transparently. The 'c'
|
|
977
|
+
function in R, is a function that concatenates its arguments making a vector.
|
|
978
|
+
|
|
979
|
+
It
|
|
980
|
+
should be clear that there is no requirement in gknit to call or use any R
|
|
981
|
+
functions. gKnit will knit standard Ruby code, or even general text without
|
|
982
|
+
any code.
|
|
983
|
+
|
|
984
|
+
```{ruby split_data}
|
|
985
|
+
a = [1, 2, 3]
|
|
986
|
+
b = "US$ 250.000"
|
|
987
|
+
c = "The 'outputs' function"
|
|
988
|
+
|
|
989
|
+
vec = R.c(1, 2, 3)
|
|
990
|
+
vec2 = R.c(10, 20, 30)
|
|
991
|
+
```
|
|
992
|
+
|
|
993
|
+
In the next block, variables 'a', 'vec' and 'vec2' are used and printed.
|
|
994
|
+
|
|
995
|
+
```{ruby split2}
|
|
996
|
+
puts a
|
|
997
|
+
puts vec * vec2
|
|
998
|
+
```
|
|
999
|
+
|
|
1000
|
+
Note that 'a' is a standard Ruby Array and 'vec' and 'vec2' are vectors that behave accordingly,
|
|
1001
|
+
where multiplication works as expected.
|
|
1002
|
+
|
|
1003
|
+
## Inline Ruby code
|
|
1004
|
+
|
|
1005
|
+
When using a Ruby chunk, the code and the output are formatted in blocks as seen above.
|
|
1006
|
+
This formatting is not always desired. Sometimes, we want to have the results of the
|
|
1007
|
+
Ruby evaluation included in the middle of a phrase. gKnit allows adding inline Ruby code
|
|
1008
|
+
with the 'rb' engine. The following chunk specification will
|
|
1009
|
+
create and inline Ruby text:
|
|
1010
|
+
|
|
1011
|
+
````
|
|
1012
|
+
This is some text with inline Ruby accessing variable 'b' which has value:
|
|
1013
|
+
```{rb puts "```{rb puts b}\n```"}
|
|
1014
|
+
```
|
|
1015
|
+
and is followed by some other text!
|
|
1016
|
+
````
|
|
1017
|
+
|
|
1018
|
+
<div style="margin-bottom:30px;">
|
|
1019
|
+
</div>
|
|
1020
|
+
|
|
1021
|
+
This is some text with inline Ruby accessing variable 'b' which has value:
|
|
1022
|
+
```{rb puts b}
|
|
1023
|
+
```
|
|
1024
|
+
and is followed by some other text!
|
|
1025
|
+
|
|
1026
|
+
<div style="margin-bottom:30px;">
|
|
1027
|
+
</div>
|
|
1028
|
+
|
|
1029
|
+
Note that it is important not to add any new line before of after the code
|
|
1030
|
+
block if we want everything to be in only one line, resulting in the following sentence
|
|
1031
|
+
with inline Ruby code.
|
|
1032
|
+
|
|
1033
|
+
|
|
1034
|
+
```{ruby heading, echo = FALSE}
|
|
1035
|
+
outputs "### #{c}"
|
|
1036
|
+
```
|
|
1037
|
+
|
|
1038
|
+
He have previously used the standard 'puts' method in Ruby chunks in order produce
|
|
1039
|
+
output. The result of a 'puts', as seen in all previous chunks that use it, is formatted
|
|
1040
|
+
inside a white box that
|
|
1041
|
+
follows the code block. Many times however, we would like to do some processing in the
|
|
1042
|
+
Ruby chunk and have the result of this processing generate and output that is
|
|
1043
|
+
"included" in the document as if we had typed it in __R markdown__ document.
|
|
1044
|
+
|
|
1045
|
+
For example, suppose we want to create a new heading in our document, but the heading
|
|
1046
|
+
phrase is the result of some code processing: maybe it's the first line of a file we are
|
|
1047
|
+
going to read. Method 'outputs' adds its output as if typed in the __R markdown__ document.
|
|
1048
|
+
|
|
1049
|
+
Take now a look at variable 'c' (it was defined in a previous block above) as
|
|
1050
|
+
'c = "The 'outputs' function". "The 'outputs' function" is actually the name of this
|
|
1051
|
+
section and it was created using the 'outputs' function inside a Ruby chunk.
|
|
1052
|
+
|
|
1053
|
+
The ruby chunk to generate this heading is:
|
|
1054
|
+
|
|
1055
|
+
````
|
|
1056
|
+
```{ruby heading}`r ''`
|
|
1057
|
+
outputs "### #{c}"
|
|
1058
|
+
```
|
|
1059
|
+
````
|
|
1060
|
+
|
|
1061
|
+
The three '###' is the way we add a Heading 3 in __R markdown__.
|
|
1062
|
+
|
|
1063
|
+
|
|
1064
|
+
### HTML Output from Ruby Chunks
|
|
1065
|
+
|
|
1066
|
+
We've just seen the use of method 'outputs' to add text to the the __R markdown__
|
|
1067
|
+
document. This technique can also be used to add HTML code to the document. In
|
|
1068
|
+
__R markdown__, any html code typed directly in the document will be properly rendered.
|
|
1069
|
+
Here, for instance, is a table definition in HTML and its output in the document:
|
|
1070
|
+
|
|
1071
|
+
```
|
|
1072
|
+
<table style="width:100%">
|
|
1073
|
+
<tr>
|
|
1074
|
+
<th>Firstname</th>
|
|
1075
|
+
<th>Lastname</th>
|
|
1076
|
+
<th>Age</th>
|
|
1077
|
+
</tr>
|
|
1078
|
+
<tr>
|
|
1079
|
+
<td>Jill</td>
|
|
1080
|
+
<td>Smith</td>
|
|
1081
|
+
<td>50</td>
|
|
1082
|
+
</tr>
|
|
1083
|
+
<tr>
|
|
1084
|
+
<td>Eve</td>
|
|
1085
|
+
<td>Jackson</td>
|
|
1086
|
+
<td>94</td>
|
|
1087
|
+
</tr>
|
|
1088
|
+
</table>
|
|
1089
|
+
```
|
|
1090
|
+
<div style="margin-bottom:30px;">
|
|
1091
|
+
</div>
|
|
1092
|
+
|
|
1093
|
+
<table style="width:100%">
|
|
1094
|
+
<tr>
|
|
1095
|
+
<th>Firstname</th>
|
|
1096
|
+
<th>Lastname</th>
|
|
1097
|
+
<th>Age</th>
|
|
1098
|
+
</tr>
|
|
1099
|
+
<tr>
|
|
1100
|
+
<td>Jill</td>
|
|
1101
|
+
<td>Smith</td>
|
|
1102
|
+
<td>50</td>
|
|
1103
|
+
</tr>
|
|
1104
|
+
<tr>
|
|
1105
|
+
<td>Eve</td>
|
|
1106
|
+
<td>Jackson</td>
|
|
1107
|
+
<td>94</td>
|
|
1108
|
+
</tr>
|
|
1109
|
+
</table>
|
|
1110
|
+
|
|
1111
|
+
<div style="margin-bottom:30px;">
|
|
1112
|
+
</div>
|
|
1113
|
+
|
|
1114
|
+
But manually creating HTML output is not always easy or desirable, specially
|
|
1115
|
+
if we intend the document to be rendered in other formats, for example, as LaTeX.
|
|
1116
|
+
Also, The above
|
|
1117
|
+
table looks ugly. The 'kableExtra' library is a great library for
|
|
1118
|
+
creating beautiful tables. Take a look at https://cran.r-project.org/web/packages/kableExtra/vignettes/awesome_table_in_html.html
|
|
1119
|
+
|
|
1120
|
+
In the next chunk, we output the 'mtcars' dataframe from R in a nicely formatted
|
|
1121
|
+
table. Note that we retrieve the mtcars dataframe by using '~R[:mtcars]'.
|
|
1122
|
+
|
|
1123
|
+
```{ruby nice_table}
|
|
1124
|
+
R.install_and_loads('kableExtra')
|
|
1125
|
+
outputs (~R[:mtcars]).kable.kable_styling
|
|
1126
|
+
```
|
|
1127
|
+
|
|
1128
|
+
## Including Ruby files in a chunk
|
|
1129
|
+
|
|
1130
|
+
R is a language that was created to be easy and fast for statisticians to use. As far
|
|
1131
|
+
as I know, it was not a
|
|
1132
|
+
language to be used for developing large systems. Of course, there are large systems and
|
|
1133
|
+
libraries in R, but the focus of the language is for developing statistical models and
|
|
1134
|
+
distribute that to peers.
|
|
1135
|
+
|
|
1136
|
+
Ruby on the other hand, is a language for large software development. Systems written in
|
|
1137
|
+
Ruby will have dozens, hundreds or even thousands of files. To document a
|
|
1138
|
+
large system with literate programming, we cannot expect the developer to add all the
|
|
1139
|
+
files in a single '.Rmd' file. gKnit provides the 'include' chunk engine to include
|
|
1140
|
+
a Ruby file as if it had being typed in the '.Rmd' file.
|
|
1141
|
+
|
|
1142
|
+
To include a file, the following chunk should be created, where <filename> is the name of
|
|
1143
|
+
the file to be included and where the extension, if it is '.rb', does not need to be added.
|
|
1144
|
+
If the 'relative' option is not included, then it is treated as TRUE. When 'relative' is
|
|
1145
|
+
true, ruby's 'require\_relative' semantics is used to load the file, when false, Ruby's
|
|
1146
|
+
\$LOAD_PATH is searched to find the file and it is 'require'd.
|
|
1147
|
+
|
|
1148
|
+
````
|
|
1149
|
+
```{include <filename>, relative = <TRUE/FALSE>}`r ''`
|
|
1150
|
+
```
|
|
1151
|
+
````
|
|
1152
|
+
|
|
1153
|
+
Below we include file 'model.rb', which is in the same directory of this blog.
|
|
1154
|
+
This code uses R 'caret' package to split a dataset in a train and test sets.
|
|
1155
|
+
The 'caret' package is a very important a useful package for doing Data Analysis,
|
|
1156
|
+
it has hundreds of functions for all steps of the Data Analysis workflow. To
|
|
1157
|
+
use 'caret' just to split a dataset is like using the proverbial cannon to
|
|
1158
|
+
kill the fly. We use it here only to show that integrating Ruby and R and
|
|
1159
|
+
using even a very complex package as 'caret' is trivial with Galaaz.
|
|
1160
|
+
|
|
1161
|
+
A word of advice: the 'caret' package has lots of dependencies and installing
|
|
1162
|
+
it in a Linux system is a time consuming operation. Method 'R.install_and_loads'
|
|
1163
|
+
will install the package if it is not already installed and can take a while.
|
|
1164
|
+
|
|
1165
|
+
````
|
|
1166
|
+
```{include model}`r ''`
|
|
1167
|
+
```
|
|
1168
|
+
````
|
|
1169
|
+
|
|
1170
|
+
```{include model}
|
|
1171
|
+
```
|
|
1172
|
+
|
|
1173
|
+
```{ruby model_partition}
|
|
1174
|
+
mtcars = ~R[:mtcars]
|
|
1175
|
+
model = Model.new(mtcars, percent_train: 0.8)
|
|
1176
|
+
model.partition(:mpg)
|
|
1177
|
+
puts model.train.head
|
|
1178
|
+
puts model.test.head
|
|
1179
|
+
```
|
|
1180
|
+
|
|
1181
|
+
## Documenting Gems
|
|
1182
|
+
|
|
1183
|
+
gKnit also allows developers to document and load files that are not in the same directory
|
|
1184
|
+
of the '.Rmd' file.
|
|
1185
|
+
|
|
1186
|
+
Here is an example of loading Ruby’s standard library file `find.rb`. In this example, relative
|
|
1187
|
+
is set to FALSE, so Ruby will look for the file in its `$LOAD_PATH`, and the user does not
|
|
1188
|
+
need to know its directory on disk.
|
|
1189
|
+
|
|
1190
|
+
````
|
|
1191
|
+
```{include find, relative = FALSE}`r ''`
|
|
1192
|
+
```
|
|
1193
|
+
````
|
|
1194
|
+
|
|
1195
|
+
```{include find, relative = FALSE}
|
|
1196
|
+
```
|
|
1197
|
+
|
|
1198
|
+
## Converting to PDF
|
|
1199
|
+
|
|
1200
|
+
One of the beauties of knitr is that the same input can be converted to many different outputs.
|
|
1201
|
+
One very useful format, is, of course, PDF. In order to converted an __R markdown__ file to PDF
|
|
1202
|
+
it is necessary to have LaTeX installed on the system. We will not explain here how to
|
|
1203
|
+
install LaTeX as there are plenty of documents on the web showing how to proceed.
|
|
1204
|
+
|
|
1205
|
+
gKnit comes with a simple LaTeX style file for gknitting this blog as a PDF document. Here is
|
|
1206
|
+
the Yaml header to generate this blog in PDF format instead of HTML:
|
|
1207
|
+
|
|
1208
|
+
```
|
|
1209
|
+
---
|
|
1210
|
+
title: "gKnit - Ruby and R Knitting with Galaaz"
|
|
1211
|
+
author: "Rodrigo Botafogo"
|
|
1212
|
+
tags: [Galaaz, Ruby, R, JRuby, knitr, gknit]
|
|
1213
|
+
date: "29 October 2018"
|
|
1214
|
+
output:
|
|
1215
|
+
pdf\_document:
|
|
1216
|
+
includes:
|
|
1217
|
+
in\_header: ["../../sty/galaaz.sty"]
|
|
1218
|
+
number\_sections: yes
|
|
1219
|
+
---
|
|
1220
|
+
```
|
|
1221
|
+
|
|
1222
|
+
## Template based documents generation
|
|
1223
|
+
|
|
1224
|
+
When a document is converted to PDF it follows a certain conversion template. We've seen above
|
|
1225
|
+
the use of 'galaaz.sty' as a basic template to generate a PDF document. Using the
|
|
1226
|
+
'gknit-draft' app that comes with Galaaz, the same .Rmd file can be compiled to different
|
|
1227
|
+
looking PDF documents. Galaaz automatically loads the 'rticles' R package that comes with
|
|
1228
|
+
templates for the following journals with the respective template name:
|
|
1229
|
+
|
|
1230
|
+
* ACM articles: acm_article
|
|
1231
|
+
* ACS articles: acs_article
|
|
1232
|
+
* AEA journal submissions: aea_article
|
|
1233
|
+
* AGU journal submissions: ????
|
|
1234
|
+
* AMS articles: ams_article
|
|
1235
|
+
* American Statistical Association: asa_article
|
|
1236
|
+
* Biometrics articles: biometrics_article
|
|
1237
|
+
* Bulletin de l'AMQ journal submissions: amq_article
|
|
1238
|
+
* CTeX documents: ctex
|
|
1239
|
+
* Elsevier journal submissions: elsevier_article
|
|
1240
|
+
* IEEE Transaction journal submissions: ieee_article
|
|
1241
|
+
* JSS articles: jss_article
|
|
1242
|
+
* MDPI journal submissions: mdpi_article
|
|
1243
|
+
* Monthly Notices of the Royal Astronomical Society articles: mnras_article
|
|
1244
|
+
* NNRAS journal submissions: nmras_article
|
|
1245
|
+
* PeerJ articles: peerj_article
|
|
1246
|
+
* Royal Society Open Science journal submissions: rsos_article
|
|
1247
|
+
* Royal Statistical Society: rss_article
|
|
1248
|
+
* Sage journal submissions: sage_article
|
|
1249
|
+
* Springer journal submissions: springer_article
|
|
1250
|
+
* Statistics in Medicine journal submissions: sim_article
|
|
1251
|
+
* Copernicus Publications journal submissions: copernicus_article
|
|
1252
|
+
* The R Journal articles: rjournal_article
|
|
1253
|
+
* Frontiers articles: ???
|
|
1254
|
+
* Taylor & Francis articles: ???
|
|
1255
|
+
* Bulletin De L'AMQ: amq_article
|
|
1256
|
+
* PLOS journal: plos_article
|
|
1257
|
+
* Proceedings of the National Academy of Sciences of the USA: pnas_article
|
|
1258
|
+
|
|
1259
|
+
In order to create a document with one of those templates, use the following command:
|
|
1260
|
+
|
|
1261
|
+
```
|
|
1262
|
+
gknit-draft --filename <my_document> --template <template> --package <package>
|
|
1263
|
+
--create_dir
|
|
1264
|
+
```
|
|
1265
|
+
So, in order to create a template for writing an R Journal, use:
|
|
1266
|
+
|
|
1267
|
+
```
|
|
1268
|
+
gknit-draft --filename my_r_article --template rjournal_article --package rticles
|
|
1269
|
+
--create_dir
|
|
1270
|
+
```
|
|
1271
|
+
|
|
1272
|
+
# Accessing R variables
|
|
1273
|
+
|
|
1274
|
+
Galaaz allows Ruby to access variables created in R. For example, the `mtcars` data set is
|
|
1275
|
+
available in R and can be accessed from Ruby by using the tilde operator followed by the
|
|
1276
|
+
symbol for the variable, in this case `:mtcars`. In the code below, method `outputs` is
|
|
1277
|
+
used to output the `mtcars` data set nicely formatted in HTML by use of the `kable` and
|
|
1278
|
+
`kable_styling` functions. Method `outputs` is only available when used with gKnit.
|
|
1279
|
+
|
|
1280
|
+
```{ruby view_kable}
|
|
1281
|
+
outputs (~R[:mtcars]).kable.kable_styling
|
|
1282
|
+
```
|
|
1283
|
+
|
|
1284
|
+
# Basic Data Types
|
|
1285
|
+
|
|
1286
|
+
## Vector
|
|
107
1287
|
|
|
108
1288
|
Vectors can be thought of as contiguous cells containing data. Cells are accessed through
|
|
109
1289
|
indexing operations such as x[5]. Galaaz has six basic (‘atomic’) vector types: logical,
|
|
@@ -116,7 +1296,7 @@ table.
|
|
|
116
1296
|
| logical | logical | logical |
|
|
117
1297
|
| integer | numeric | integer |
|
|
118
1298
|
| double | numeric | double |
|
|
119
|
-
| complex | complex |
|
|
1299
|
+
| complex | complex | complex |
|
|
120
1300
|
| character | character | character |
|
|
121
1301
|
| raw | raw | raw |
|
|
122
1302
|
|
|
@@ -178,7 +1358,7 @@ vec = R.c(true, true, false, false, true)
|
|
|
178
1358
|
puts vec
|
|
179
1359
|
```
|
|
180
1360
|
|
|
181
|
-
|
|
1361
|
+
### Combining Vectors
|
|
182
1362
|
|
|
183
1363
|
The 'c' functions used to create vectors can also be used to combine two vectors:
|
|
184
1364
|
|
|
@@ -193,14 +1373,14 @@ In this next example, method 'c' is chainned after 'vec1'. This also looks like
|
|
|
193
1373
|
method of the vector, but in reallity, this is actually closer to the pipe operator. When
|
|
194
1374
|
Galaaz identifies that 'c' is not a method of 'vec' it actually tries to call 'R.c' with
|
|
195
1375
|
'vec1' as the first argument concatenated with all the other available arguments. The code
|
|
196
|
-
|
|
1376
|
+
below is automatically converted to the code above.
|
|
197
1377
|
|
|
198
1378
|
```{ruby chainning_methods}
|
|
199
1379
|
vec = vec1.c(vec2)
|
|
200
1380
|
puts vec
|
|
201
1381
|
```
|
|
202
1382
|
|
|
203
|
-
|
|
1383
|
+
### Vector Arithmetic
|
|
204
1384
|
|
|
205
1385
|
Arithmetic operations on vectors are performed element by element:
|
|
206
1386
|
|
|
@@ -219,7 +1399,7 @@ vec3 = R.c(1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0)
|
|
|
219
1399
|
puts vec4 = vec1 + vec3
|
|
220
1400
|
```
|
|
221
1401
|
|
|
222
|
-
|
|
1402
|
+
### Vector Indexing
|
|
223
1403
|
|
|
224
1404
|
Vectors can be indexed by using the '[]' operator:
|
|
225
1405
|
|
|
@@ -227,7 +1407,7 @@ Vectors can be indexed by using the '[]' operator:
|
|
|
227
1407
|
puts vec4[3]
|
|
228
1408
|
```
|
|
229
1409
|
|
|
230
|
-
We can also index a vector with another vector. For example, in the code
|
|
1410
|
+
We can also index a vector with another vector. For example, in the code below, we take elements
|
|
231
1411
|
1, 3, 5, and 7 from vec3:
|
|
232
1412
|
|
|
233
1413
|
```{ruby index_by_vector}
|
|
@@ -275,7 +1455,7 @@ full_name = R.c(First: "Rodrigo", Middle: "A", Last: "Botafogo")
|
|
|
275
1455
|
puts full_name
|
|
276
1456
|
```
|
|
277
1457
|
|
|
278
|
-
|
|
1458
|
+
### Extracting Native Ruby Types from a Vector
|
|
279
1459
|
|
|
280
1460
|
Vectors created with 'R.c' are of class R::Vector. You might have noticed that when indexing a
|
|
281
1461
|
vector, a new vector is returned, even if this vector has one single element. In order to use
|
|
@@ -290,19 +1470,7 @@ puts vec4 >> 4
|
|
|
290
1470
|
|
|
291
1471
|
Note that indexing with '>>' starts at 0 and not at 1, also, we cannot do negative indexing.
|
|
292
1472
|
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
Galaaz allows Ruby to access variables created in R. For example, the 'mtcars' data set is
|
|
296
|
-
available in R and can be accessed from Ruby by using the 'tilda' operator followed by the
|
|
297
|
-
symbol for the variable, in this case ':mtcar'. In the code bellow method 'outputs' is
|
|
298
|
-
used to output the 'mtcars' data set nicely formatted in HTML by use of the 'kable' and
|
|
299
|
-
'kable_styling' functions. Method 'outputs' is only available when used with 'gknit'.
|
|
300
|
-
|
|
301
|
-
```{ruby view_kable}
|
|
302
|
-
outputs (~:mtcars).kable.kable_styling
|
|
303
|
-
```
|
|
304
|
-
|
|
305
|
-
# Matrix
|
|
1473
|
+
## Matrix
|
|
306
1474
|
|
|
307
1475
|
A matrix is a collection of elements organized as a two dimensional table. A matrix can be
|
|
308
1476
|
created by the 'matrix' function:
|
|
@@ -326,7 +1494,7 @@ mat_row = R.matrix(R.c(1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0),
|
|
|
326
1494
|
puts mat_row
|
|
327
1495
|
```
|
|
328
1496
|
|
|
329
|
-
|
|
1497
|
+
### Indexing a Matrix
|
|
330
1498
|
|
|
331
1499
|
A matrix can be indexed by [row, column]:
|
|
332
1500
|
|
|
@@ -360,7 +1528,7 @@ and 'cbind':
|
|
|
360
1528
|
puts mat_row.cbind(mat)
|
|
361
1529
|
```
|
|
362
1530
|
|
|
363
|
-
|
|
1531
|
+
## List
|
|
364
1532
|
|
|
365
1533
|
A list is a data structure that can contain sublists of different types, while vector and matrix
|
|
366
1534
|
can only hold one type of element.
|
|
@@ -376,7 +1544,7 @@ puts lst
|
|
|
376
1544
|
Note that 'lst' elements are named elements.
|
|
377
1545
|
|
|
378
1546
|
|
|
379
|
-
|
|
1547
|
+
### List Indexing
|
|
380
1548
|
|
|
381
1549
|
List indexing, also called slicing, is done using the '[]' operator and the '[[]]' operator. Let's
|
|
382
1550
|
first start with the '[]' operator. The list above has three sublist indexing with '[]' will
|
|
@@ -406,11 +1574,11 @@ then the first element of the vector was extracted (note that vectors also accep
|
|
|
406
1574
|
operator) and then the vector was indexed by its first element, extracting the native Ruby type.
|
|
407
1575
|
|
|
408
1576
|
|
|
409
|
-
|
|
1577
|
+
## Data Frame
|
|
410
1578
|
|
|
411
1579
|
A data frame is a table like structure in which each column has the same number of
|
|
412
1580
|
rows. Data frames are the basic structure for storing data for data analysis. We have already
|
|
413
|
-
seen a data frame previously when we accessed variable '
|
|
1581
|
+
seen a data frame previously when we accessed variable '~R[:mtcars]'. In order to create a
|
|
414
1582
|
data frame, function 'data__frame' is used:
|
|
415
1583
|
|
|
416
1584
|
```{ruby dataframe}
|
|
@@ -421,51 +1589,51 @@ df = R.data__frame(
|
|
|
421
1589
|
puts df
|
|
422
1590
|
```
|
|
423
1591
|
|
|
424
|
-
|
|
1592
|
+
### Data Frame Indexing
|
|
425
1593
|
|
|
426
1594
|
A data frame can be indexed the same way as a matrix, by using '[row, column]', where row and
|
|
427
1595
|
column can either be a numeric or the name of the row or column
|
|
428
1596
|
|
|
429
1597
|
```{ruby dataframe_index}
|
|
430
|
-
puts (
|
|
431
|
-
puts (
|
|
432
|
-
puts (
|
|
1598
|
+
puts (~R[:mtcars]).head
|
|
1599
|
+
puts (~R[:mtcars])[1, 2]
|
|
1600
|
+
puts (~R[:mtcars])['Datsun 710', 'mpg']
|
|
433
1601
|
```
|
|
434
1602
|
|
|
435
1603
|
Extracting a column from a data frame as a vector can be done by using the double square bracket
|
|
436
1604
|
operator:
|
|
437
1605
|
|
|
438
1606
|
```{ruby dataframe_column}
|
|
439
|
-
puts (
|
|
1607
|
+
puts (~R[:mtcars])[['mpg']]
|
|
440
1608
|
```
|
|
441
1609
|
|
|
442
1610
|
A data frame column can also be accessed as if it were an instance variable of the data frame:
|
|
443
1611
|
|
|
444
1612
|
```{ruby dataframe_instance_variable}
|
|
445
|
-
puts (
|
|
1613
|
+
puts (~R[:mtcars]).mpg
|
|
446
1614
|
```
|
|
447
1615
|
|
|
448
1616
|
Slicing a data frame can be done by indexing it with a vector (we use 'head' to reduce the
|
|
449
1617
|
output):
|
|
450
1618
|
|
|
451
1619
|
```{ruby dataframe_column_slice}
|
|
452
|
-
puts (
|
|
1620
|
+
puts (~R[:mtcars])[R.c('mpg', 'hp')].head
|
|
453
1621
|
```
|
|
454
1622
|
|
|
455
1623
|
A row slice can be obtained by indexing by row and using the ':all' keyword for the column:
|
|
456
1624
|
|
|
457
1625
|
```{ruby dataframe_row_slice}
|
|
458
|
-
puts (
|
|
1626
|
+
puts (~R[:mtcars])[R.c('Datsun 710', 'Camaro Z28'), :all]
|
|
459
1627
|
```
|
|
460
1628
|
|
|
461
1629
|
Finally, a data frame can also be indexed with a logical vector. In this next example, the
|
|
462
1630
|
'am' column of :mtcars is compared with 0 (with method 'eq'). When 'am' is equal to 0 the
|
|
463
|
-
car is automatic. So, by doing '(
|
|
1631
|
+
car is automatic. So, by doing '(~R[:mtcars]).am.eq 0' a logical vector is created with
|
|
464
1632
|
'true' whenever 'am' is 0 and 'false' otherwise.
|
|
465
1633
|
|
|
466
1634
|
```{ruby logical_vector_filter}
|
|
467
1635
|
# obtain a vector with 'true' for cars with automatic transmission
|
|
468
|
-
automatic = (
|
|
1636
|
+
automatic = (~R[:mtcars]).am.eq 0
|
|
469
1637
|
puts automatic
|
|
470
1638
|
```
|
|
471
1639
|
|
|
@@ -474,7 +1642,7 @@ which all cars have automatic transmission.
|
|
|
474
1642
|
|
|
475
1643
|
```{ruby dataframe_logical}
|
|
476
1644
|
# slice the data frame by using this vector
|
|
477
|
-
puts (
|
|
1645
|
+
puts (~R[:mtcars])[automatic, :all]
|
|
478
1646
|
```
|
|
479
1647
|
|
|
480
1648
|
# Writing Expressions in Galaaz
|
|
@@ -484,24 +1652,24 @@ Galaaz extends Ruby to work with complex expressions, similar to R's expressions
|
|
|
484
1652
|
|
|
485
1653
|
## Expressions from operators
|
|
486
1654
|
|
|
487
|
-
The code
|
|
1655
|
+
The code below
|
|
488
1656
|
creates an expression summing two symbols
|
|
489
1657
|
|
|
490
1658
|
```{ruby expressions}
|
|
491
|
-
exp1 = :a + :b
|
|
1659
|
+
exp1 = R[:a] + R[:b]
|
|
492
1660
|
puts exp1
|
|
493
1661
|
```
|
|
494
1662
|
We can build any complex mathematical expression
|
|
495
1663
|
|
|
496
1664
|
```{ruby expr2}
|
|
497
|
-
exp2 = (:a + :b) * 2.0 + :c ** 2 / :z
|
|
1665
|
+
exp2 = (R[:a] + R[:b]) * 2.0 + R[:c] ** 2 / R[:z]
|
|
498
1666
|
puts exp2
|
|
499
1667
|
```
|
|
500
1668
|
|
|
501
1669
|
It is also possible to use inequality operators in building expressions
|
|
502
1670
|
|
|
503
1671
|
```{ruby expr3}
|
|
504
|
-
exp3 = (:a + :b) >= :z
|
|
1672
|
+
exp3 = (R[:a] + R[:b]) >= :z
|
|
505
1673
|
puts exp3
|
|
506
1674
|
```
|
|
507
1675
|
|
|
@@ -510,7 +1678,7 @@ notation for those operators such as (.gt, .ge, etc.). So the same expression w
|
|
|
510
1678
|
above can also be written as
|
|
511
1679
|
|
|
512
1680
|
```{ruby expr4}
|
|
513
|
-
exp4 = (:a + :b).ge :z
|
|
1681
|
+
exp4 = (R[:a] + R[:b]).ge :z
|
|
514
1682
|
puts exp4
|
|
515
1683
|
```
|
|
516
1684
|
|
|
@@ -519,23 +1687,23 @@ those are expressions involving '==', and '='. In order to write an expression
|
|
|
519
1687
|
need to use the method '.eq' and for '=' we need the function '.assign'
|
|
520
1688
|
|
|
521
1689
|
```{ruby expr5}
|
|
522
|
-
exp5 = (:a + :b).eq :z
|
|
1690
|
+
exp5 = (R[:a] + R[:b]).eq :z
|
|
523
1691
|
puts exp5
|
|
524
1692
|
```
|
|
525
1693
|
|
|
526
1694
|
```{ruby expr6}
|
|
527
|
-
exp6 = :y.assign :a + :b
|
|
1695
|
+
exp6 = R[:y].assign R[:a] + R[:b]
|
|
528
1696
|
puts exp6
|
|
529
1697
|
```
|
|
530
1698
|
In general we think that using the functional notation is preferable to using the
|
|
531
1699
|
symbolic notation as otherwise, we end up writing invalid expressions such as
|
|
532
1700
|
|
|
533
|
-
```{ruby exp_wrong, warning=FALSE}
|
|
534
|
-
exp_wrong = (:a + :b) == :z
|
|
1701
|
+
```{ruby exp_wrong, warning=FALSE, eval=FALSE}
|
|
1702
|
+
exp_wrong = (R[:a] + R[:b]) == :z
|
|
535
1703
|
puts exp_wrong
|
|
536
1704
|
```
|
|
537
1705
|
and it might be difficult to understand what is going on here. The problem lies with the fact that
|
|
538
|
-
when using '==' we are comparing expression (:a + :b) to expression :z with '=='. When the
|
|
1706
|
+
when using '==' we are comparing expression (R[:a] + R[:b]) to expression :z with '=='. When the
|
|
539
1707
|
comparison is executed, the system tries to evaluate :a, :b and :z, and those symbols at
|
|
540
1708
|
this time are not bound to anything and we get a "object 'a' not found" message.
|
|
541
1709
|
If we only use functional notation, this type of error will not occur.
|
|
@@ -550,21 +1718,21 @@ When we want the function to be part of the expression, we call the function pre
|
|
|
550
1718
|
by the letter E, such as 'E.sin(x)'
|
|
551
1719
|
|
|
552
1720
|
```{ruby method_expression}
|
|
553
|
-
exp7 = :y.assign E.sin(:x)
|
|
1721
|
+
exp7 = R[:y].assign E.sin(R[:x])
|
|
554
1722
|
puts exp7
|
|
555
1723
|
```
|
|
556
1724
|
|
|
557
1725
|
Expressions can also be written using '.' notation:
|
|
558
1726
|
|
|
559
1727
|
```{ruby expression_with_dot}
|
|
560
|
-
exp8 = :y.assign :x.sin
|
|
1728
|
+
exp8 = R[:y].assign R[:x].sin
|
|
561
1729
|
puts exp8
|
|
562
1730
|
```
|
|
563
1731
|
|
|
564
1732
|
When a function has multiple arguments, the first one can be used before the '.':
|
|
565
1733
|
|
|
566
1734
|
```{ruby expression_multiple_args}
|
|
567
|
-
exp9 = :x.c(:y)
|
|
1735
|
+
exp9 = R[:x].c(R[:y])
|
|
568
1736
|
puts exp9
|
|
569
1737
|
```
|
|
570
1738
|
|
|
@@ -574,7 +1742,7 @@ Expressions can be evaluated by calling function 'eval' with a binding. A bindin
|
|
|
574
1742
|
with a list:
|
|
575
1743
|
|
|
576
1744
|
```{ruby eval_expression_list}
|
|
577
|
-
exp = (:a + :b) * 2.0 + :c ** 2 / :z
|
|
1745
|
+
exp = (R[:a] + R[:b]) * 2.0 + R[:c] ** 2 / R[:z]
|
|
578
1746
|
puts exp.eval(R.list(a: 10, b: 20, c: 30, z: 40))
|
|
579
1747
|
```
|
|
580
1748
|
|
|
@@ -593,35 +1761,39 @@ puts exp.eval(df)
|
|
|
593
1761
|
# Manipulating Data
|
|
594
1762
|
|
|
595
1763
|
One of the major benefits of Galaaz is to bring strong data manipulation to Ruby. The following
|
|
596
|
-
examples were extracted from
|
|
1764
|
+
examples were extracted from Hadley's "R for Data Science" (https://r4ds.had.co.nz/). This
|
|
597
1765
|
is a highly recommended book for those not already familiar with the 'tidyverse' style of
|
|
598
1766
|
programming in R. In the sections to follow, we will limit ourselves to convert the R code to
|
|
599
1767
|
Galaaz.
|
|
600
1768
|
|
|
601
1769
|
For these
|
|
602
1770
|
examples, we will investigate the nycflights13 data set available on the package by the
|
|
603
|
-
same name. We use function 'R.
|
|
1771
|
+
same name. We use function 'R.install\_and\_loads' that checks if the library is available
|
|
604
1772
|
locally, and if not, installs it. This data frame contains all 336,776 flights that
|
|
605
1773
|
departed from New York City in 2013. The data comes from the US Bureau of
|
|
606
1774
|
Transportation Statistics.
|
|
607
1775
|
|
|
1776
|
+
Dplyr often uses **tibbles** in place of classic data frames. In Galaaz, printing may differ from
|
|
1777
|
+
the R console; if you need a classic tabular printout, convert with **`as__data__frame`** (or use
|
|
1778
|
+
`head` / `str` in R via `R` calls).
|
|
1779
|
+
|
|
608
1780
|
```{ruby nycflights13}
|
|
609
1781
|
R.install_and_loads('nycflights13')
|
|
610
1782
|
R.library('dplyr')
|
|
611
1783
|
```
|
|
612
1784
|
|
|
613
1785
|
```{ruby flights}
|
|
614
|
-
flights =
|
|
615
|
-
puts flights.head
|
|
1786
|
+
flights = ~R[:flights]
|
|
1787
|
+
puts flights.head
|
|
616
1788
|
```
|
|
617
1789
|
|
|
618
1790
|
## Filtering rows with Filter
|
|
619
1791
|
|
|
620
1792
|
In this example we filter the flights data set by giving to the filter function two expressions:
|
|
621
|
-
the first :month.eq 1
|
|
1793
|
+
the first R[:month].eq 1
|
|
622
1794
|
|
|
623
1795
|
```{ruby filter_rows}
|
|
624
|
-
puts flights.filter((:month.eq 1), (:day.eq 1)).head
|
|
1796
|
+
puts flights.filter((R[:month].eq 1), (R[:day].eq 1)).head
|
|
625
1797
|
```
|
|
626
1798
|
|
|
627
1799
|
## Logical Operators
|
|
@@ -629,7 +1801,7 @@ puts flights.filter((:month.eq 1), (:day.eq 1)).head.as__data__frame
|
|
|
629
1801
|
All flights that departed in November of December
|
|
630
1802
|
|
|
631
1803
|
```{ruby nov_dec}
|
|
632
|
-
puts flights.filter((:month.eq 11) | (:month.eq 12)).head
|
|
1804
|
+
puts flights.filter((R[:month].eq 11) | (R[:month].eq 12)).head
|
|
633
1805
|
```
|
|
634
1806
|
|
|
635
1807
|
The same as above, but using the 'in' operator. In R, it is possible to define many operators
|
|
@@ -638,7 +1810,7 @@ operators from Galaaz the '._' method is used, where the first argument is the o
|
|
|
638
1810
|
symbol, in this case ':in' and the second argument is the vector:
|
|
639
1811
|
|
|
640
1812
|
```{ruby in_op}
|
|
641
|
-
puts flights.filter(:month._ :in, R.c(11, 12)).head
|
|
1813
|
+
puts flights.filter(R[:month]._ :in, R.c(11, 12)).head
|
|
642
1814
|
```
|
|
643
1815
|
|
|
644
1816
|
## Filtering with NA (Not Available)
|
|
@@ -650,20 +1822,20 @@ what is obtained from data frame.
|
|
|
650
1822
|
|
|
651
1823
|
```{ruby na_tibble}
|
|
652
1824
|
df = R.tibble(x: R.c(1, R::NA, 3))
|
|
653
|
-
puts df
|
|
1825
|
+
puts df
|
|
654
1826
|
```
|
|
655
1827
|
|
|
656
1828
|
Now filtering by :x > 1 shows all lines that satisfy this condition, where the row with R:NA does
|
|
657
1829
|
not.
|
|
658
1830
|
|
|
659
1831
|
```{ruby filter_na}
|
|
660
|
-
puts df.filter(:x > 1)
|
|
1832
|
+
puts df.filter(R[:x] > 1)
|
|
661
1833
|
```
|
|
662
1834
|
|
|
663
1835
|
To match an NA use method 'is__na'
|
|
664
1836
|
|
|
665
1837
|
```{ruby with_na}
|
|
666
|
-
puts df.filter((:x.is__na) | (:x > 1))
|
|
1838
|
+
puts df.filter((R[:x].is__na) | (R[:x] > 1))
|
|
667
1839
|
```
|
|
668
1840
|
|
|
669
1841
|
## Arrange Rows with arrange
|
|
@@ -671,13 +1843,13 @@ puts df.filter((:x.is__na) | (:x > 1)).as__data__frame
|
|
|
671
1843
|
Arrange reorders the rows of a data frame by the given arguments.
|
|
672
1844
|
|
|
673
1845
|
```{ruby arrange}
|
|
674
|
-
puts flights.arrange(:year, :month, :day).head
|
|
1846
|
+
puts flights.arrange(:year, :month, :day).head
|
|
675
1847
|
```
|
|
676
1848
|
|
|
677
1849
|
To arrange in descending order, use function 'desc'
|
|
678
1850
|
|
|
679
1851
|
```{ruby desc_arrange}
|
|
680
|
-
puts flights.arrange(:dep_delay.desc).head
|
|
1852
|
+
puts flights.arrange(R[:dep_delay].desc).head
|
|
681
1853
|
```
|
|
682
1854
|
|
|
683
1855
|
## Selecting columns
|
|
@@ -685,19 +1857,19 @@ puts flights.arrange(:dep_delay.desc).head.as__data__frame
|
|
|
685
1857
|
To select specific columns from a dataset we use function 'select':
|
|
686
1858
|
|
|
687
1859
|
```{ruby select}
|
|
688
|
-
puts flights.select(:year, :month, :day).head
|
|
1860
|
+
puts flights.select(:year, :month, :day).head
|
|
689
1861
|
```
|
|
690
1862
|
|
|
691
1863
|
It is also possible to select column in a given range
|
|
692
1864
|
|
|
693
1865
|
```{ruby select_range}
|
|
694
|
-
puts flights.select(:year.up_to
|
|
1866
|
+
puts flights.select(R[:year].up_to(R[:day])).head
|
|
695
1867
|
```
|
|
696
1868
|
|
|
697
1869
|
Select all columns that start with a given name sequence
|
|
698
1870
|
|
|
699
1871
|
```{ruby select_starts_with}
|
|
700
|
-
puts flights.select(E.starts_with('arr')).head
|
|
1872
|
+
puts flights.select(E.starts_with('arr')).head
|
|
701
1873
|
```
|
|
702
1874
|
|
|
703
1875
|
Other functions that can be used:
|
|
@@ -714,26 +1886,26 @@ Other functions that can be used:
|
|
|
714
1886
|
A helper function that comes in handy when we just want to rearrange column order is 'Everything':
|
|
715
1887
|
|
|
716
1888
|
```{ruby everything}
|
|
717
|
-
puts flights.select(:year, :month, :day, E.everything).head
|
|
1889
|
+
puts flights.select(:year, :month, :day, E.everything).head
|
|
718
1890
|
```
|
|
719
1891
|
|
|
720
1892
|
## Add variables to a dataframe with 'mutate'
|
|
721
1893
|
|
|
722
1894
|
```{ruby small_flights}
|
|
723
1895
|
flights_sm = flights.
|
|
724
|
-
select((:year.up_to
|
|
1896
|
+
select((R[:year].up_to(R[:day])),
|
|
725
1897
|
E.ends_with('delay'),
|
|
726
1898
|
:distance,
|
|
727
1899
|
:air_time)
|
|
728
1900
|
|
|
729
|
-
puts flights_sm.head
|
|
1901
|
+
puts flights_sm.head
|
|
730
1902
|
```
|
|
731
1903
|
|
|
732
1904
|
```{ruby mutate}
|
|
733
1905
|
flights_sm = flights_sm.
|
|
734
|
-
mutate(gain: :dep_delay - :arr_delay,
|
|
735
|
-
speed: :distance / :air_time * 60)
|
|
736
|
-
puts flights_sm.head
|
|
1906
|
+
mutate(gain: R[:dep_delay] - R[:arr_delay],
|
|
1907
|
+
speed: R[:distance] / R[:air_time] * 60)
|
|
1908
|
+
puts flights_sm.head
|
|
737
1909
|
```
|
|
738
1910
|
|
|
739
1911
|
## Summarising data
|
|
@@ -742,14 +1914,14 @@ Function 'summarise' calculates summaries for the data frame. When no 'group_by'
|
|
|
742
1914
|
a single value is obtained from the data frame:
|
|
743
1915
|
|
|
744
1916
|
```{ruby summarise}
|
|
745
|
-
puts flights.summarise(delay: E.mean(:dep_delay, na__rm: true))
|
|
1917
|
+
puts flights.summarise(delay: E.mean(:dep_delay, na__rm: true))
|
|
746
1918
|
```
|
|
747
1919
|
|
|
748
|
-
When a data frame is
|
|
1920
|
+
When a data frame is grouped with 'group_by' summaries apply to the given group:
|
|
749
1921
|
|
|
750
1922
|
```{ruby summarise_group_by}
|
|
751
1923
|
by_day = flights.group_by(:year, :month, :day)
|
|
752
|
-
puts by_day.summarise(delay: :dep_delay.mean(na__rm: true)).head
|
|
1924
|
+
puts by_day.summarise(delay: R[:dep_delay].mean(na__rm: true)).head
|
|
753
1925
|
```
|
|
754
1926
|
|
|
755
1927
|
Next we put many operations together by pipping them one after the other:
|
|
@@ -759,23 +1931,25 @@ delays = flights.
|
|
|
759
1931
|
group_by(:dest).
|
|
760
1932
|
summarise(
|
|
761
1933
|
count: E.n,
|
|
762
|
-
dist: :distance.mean(na__rm: true),
|
|
763
|
-
delay: :arr_delay.mean(na__rm: true)).
|
|
764
|
-
filter(:count > 20, :dest != "NHL")
|
|
1934
|
+
dist: R[:distance].mean(na__rm: true),
|
|
1935
|
+
delay: R[:arr_delay].mean(na__rm: true)).
|
|
1936
|
+
filter(R[:count] > 20, R[:dest] != "NHL")
|
|
765
1937
|
|
|
766
|
-
puts delays.
|
|
1938
|
+
puts delays.head
|
|
767
1939
|
```
|
|
768
1940
|
|
|
769
1941
|
# Using Data Table
|
|
770
1942
|
|
|
1943
|
+
The next chunk converts the **nycflights13** `flights` tibble already loaded above into a
|
|
1944
|
+
**`data.table`**. That keeps the manual offline and avoids downloading a remote CSV during gknit
|
|
1945
|
+
(network stalls look like bridge hangs when the transfer runs inside a single R eval).
|
|
1946
|
+
|
|
771
1947
|
```{ruby fread}
|
|
772
1948
|
R.library('data.table')
|
|
773
|
-
R.install_and_loads('curl')
|
|
774
1949
|
|
|
775
|
-
|
|
776
|
-
flights = R.fread(input)
|
|
777
|
-
puts flights
|
|
1950
|
+
flights = R.as__data__table(~R[:flights])
|
|
778
1951
|
puts flights.dim
|
|
1952
|
+
puts R.head(flights, 12)
|
|
779
1953
|
```
|
|
780
1954
|
|
|
781
1955
|
```{ruby data_table}
|
|
@@ -793,7 +1967,7 @@ puts data_table.ID
|
|
|
793
1967
|
|
|
794
1968
|
```{ruby subset_i}
|
|
795
1969
|
# subset rows in i
|
|
796
|
-
ans = flights[(:origin.eq "JFK") & (:month.eq 6)]
|
|
1970
|
+
ans = flights[(R[:origin].eq "JFK") & (R[:month].eq 6)]
|
|
797
1971
|
puts ans.head
|
|
798
1972
|
|
|
799
1973
|
# Get the first two rows from flights.
|
|
@@ -801,8 +1975,7 @@ puts ans.head
|
|
|
801
1975
|
ans = flights[(1..2)]
|
|
802
1976
|
puts ans
|
|
803
1977
|
|
|
804
|
-
# Sort
|
|
805
|
-
|
|
1978
|
+
# Sort by origin asc, then dest desc (example kept commented):
|
|
806
1979
|
# ans = flights[E.order(:origin, -(:dest))]
|
|
807
1980
|
# puts ans.head
|
|
808
1981
|
|
|
@@ -815,66 +1988,274 @@ puts ans
|
|
|
815
1988
|
ans = flights[:all, :arr_delay]
|
|
816
1989
|
puts ans.head
|
|
817
1990
|
|
|
818
|
-
#
|
|
1991
|
+
# arr_delay as data.table (not plain vector).
|
|
819
1992
|
|
|
820
|
-
ans = flights[:all, :arr_delay.list]
|
|
1993
|
+
ans = flights[:all, R[:arr_delay].list]
|
|
821
1994
|
puts ans.head
|
|
822
1995
|
|
|
823
|
-
ans = flights[:all, E.list(:arr_delay, :dep_delay)]
|
|
1996
|
+
ans = flights[:all, E.list(R[:arr_delay], R[:dep_delay])]
|
|
1997
|
+
```
|
|
1998
|
+
|
|
1999
|
+
# Apache Arrow
|
|
2000
|
+
|
|
2001
|
+
[Apache Arrow](https://arrow.apache.org/) is a **columnar** in-memory format used heavily in R
|
|
2002
|
+
and Python for analytics. In Galaaz, **Ruby does not hold an Arrow C++ table itself**; instead you
|
|
2003
|
+
build ordinary Ruby structures (arrays of row hashes), and **`R::Arrow.from_ruby_batches`** creates
|
|
2004
|
+
a real **Arrow `Table` inside GNU R**. From there you use R’s **`arrow`** and **`dplyr`** packages
|
|
2005
|
+
as usual: **`group_by`** on the Arrow table, **`summarise`** for aggregates, then **`collect()`** to
|
|
2006
|
+
materialize a tibble when you need in-memory R rows.
|
|
2007
|
+
|
|
2008
|
+
That pattern matches production use: **JRuby threads** (or sequential code) assemble many rows in
|
|
2009
|
+
Ruby; you pay **one** bridge-heavy handoff to R; **dplyr** runs vectorised work on the Arrow table
|
|
2010
|
+
in R.
|
|
2011
|
+
|
|
2012
|
+
**Prerequisites:** install R packages **`arrow`** and **`dplyr`**. Run scripts with
|
|
2013
|
+
**`bin/galaaz-jruby`** (or the same JVM flags as in **`docs/testing.md`**) so the Arrow JNI stack is
|
|
2014
|
+
available.
|
|
2015
|
+
|
|
2016
|
+
## Other `R::Arrow` helpers
|
|
2017
|
+
|
|
2018
|
+
The Ruby module **`R::Arrow`** (see `lib/R_interface/r_arrow.rb`) also includes:
|
|
2019
|
+
|
|
2020
|
+
* **`R::Arrow.table_from(df)`** — wrap an R `data.frame` / tibble as an Arrow table.
|
|
2021
|
+
* **`R::Arrow.read_feather` / `write_feather`**, **`read_parquet`**, **`dataset(path)`** — file and
|
|
2022
|
+
dataset IO on paths visible to R.
|
|
2023
|
+
|
|
2024
|
+
## Example: many Ruby rows → Arrow in R → grouped statistics
|
|
2025
|
+
|
|
2026
|
+
The repository test **`slow-specs/arrow_large_pipeline_spec.rb`** builds **200k rows** in parallel
|
|
2027
|
+
(eight threads × 25,000 rows), pushes them through **`R::Arrow.from_ruby_batches`**, then checks that
|
|
2028
|
+
**dplyr** group summaries match a Ruby reference calculation. The same logic appears below at a
|
|
2029
|
+
**smaller scale** so this manual can knit quickly; increase `thread_count` and `rows_per_thread`
|
|
2030
|
+
when experimenting locally.
|
|
2031
|
+
|
|
2032
|
+
```{ruby arrow_pipeline_example, message=FALSE, warning=FALSE}
|
|
2033
|
+
# Scaled-down version of slow-specs/arrow_large_pipeline_spec.rb.
|
|
2034
|
+
unless R::Support.eval("requireNamespace('arrow', quietly=TRUE) && requireNamespace('dplyr', quietly=TRUE)") == true
|
|
2035
|
+
puts '(Skip: need arrow + dplyr in R; use bin/galaaz-jruby outside gKnit.)'
|
|
2036
|
+
else
|
|
2037
|
+
thread_count = 4
|
|
2038
|
+
rows_per_thread = 500
|
|
2039
|
+
group_count = 5
|
|
2040
|
+
|
|
2041
|
+
batches = []
|
|
2042
|
+
mutex = Mutex.new
|
|
2043
|
+
threads = []
|
|
2044
|
+
|
|
2045
|
+
thread_count.times do |tid|
|
|
2046
|
+
threads << Thread.new do
|
|
2047
|
+
start = tid * rows_per_thread
|
|
2048
|
+
local = (start...(start + rows_per_thread)).map do |i|
|
|
2049
|
+
{
|
|
2050
|
+
id: i,
|
|
2051
|
+
grp: "g#{i % group_count}",
|
|
2052
|
+
value: (i % 17) + 1,
|
|
2053
|
+
weight: ((i % 5) + 1) * 0.5
|
|
2054
|
+
}
|
|
2055
|
+
end
|
|
2056
|
+
mutex.synchronize { batches << local }
|
|
2057
|
+
end
|
|
2058
|
+
end
|
|
2059
|
+
threads.each(&:join)
|
|
2060
|
+
|
|
2061
|
+
tbl = R::Arrow.from_ruby_batches(batches)
|
|
2062
|
+
puts "R class after from_ruby_batches: #{tbl.rclass}"
|
|
2063
|
+
|
|
2064
|
+
grouped = R.dplyr___group_by(tbl, :grp)
|
|
2065
|
+
summarised = R.dplyr___summarise(
|
|
2066
|
+
grouped,
|
|
2067
|
+
n: E.n(),
|
|
2068
|
+
total: E.sum(:value),
|
|
2069
|
+
wsum: E.sum(R[:value] * R[:weight])
|
|
2070
|
+
)
|
|
2071
|
+
out = R.dplyr___collect(summarised)
|
|
2072
|
+
|
|
2073
|
+
puts 'Per-group summary (first rows):'
|
|
2074
|
+
puts R.as__data__frame(out).head(10)
|
|
2075
|
+
|
|
2076
|
+
total_n = 0
|
|
2077
|
+
(1..(out.nrow >> 0)).each { |i| total_n += (out[['n']][i] >> 0) }
|
|
2078
|
+
puts "Sum of group counts n (should equal #{thread_count * rows_per_thread}): #{total_n}"
|
|
2079
|
+
end
|
|
2080
|
+
```
|
|
2081
|
+
|
|
2082
|
+
**What to notice:** (1) Ruby only sees **`Hash`** rows and Ruby **`Thread`** objects; (2) a single
|
|
2083
|
+
**`from_ruby_batches`** call creates the Arrow table in R; (3) **`dplyr___group_by`** /
|
|
2084
|
+
**`dplyr___summarise`** / **`dplyr___collect`** mirror **`dplyr::group_by`** /
|
|
2085
|
+
**`dplyr::summarise`** / **`dplyr::collect`** on an Arrow-backed table. For a lighter test, see
|
|
2086
|
+
**`specs/arrow_from_ruby_batches_spec.rb`**; for the full-size benchmark, run
|
|
2087
|
+
**`bin/run_slow_rspec slow-specs/arrow_large_pipeline_spec.rb`**.
|
|
2088
|
+
|
|
2089
|
+
# Bioconductor and DESeq2
|
|
2090
|
+
|
|
2091
|
+
**Bioconductor** packages are ordinary R packages installed from the Bioconductor repositories.
|
|
2092
|
+
Galaaz does not treat them specially: once installed in **GNU R**, you load them with
|
|
2093
|
+
**`R.library`** like any CRAN package.
|
|
2094
|
+
|
|
2095
|
+
## Installing Bioconductor packages
|
|
2096
|
+
|
|
2097
|
+
From an R session (or `R -e '...'`), use **BiocManager** (see
|
|
2098
|
+
[bioconductor.org](https://bioconductor.org/install/)):
|
|
2099
|
+
|
|
2100
|
+
```r
|
|
2101
|
+
if (!requireNamespace("BiocManager", quietly = TRUE))
|
|
2102
|
+
install.packages("BiocManager")
|
|
2103
|
+
BiocManager::install(c("DESeq2", "airway"))
|
|
2104
|
+
```
|
|
2105
|
+
|
|
2106
|
+
The **`airway`** package ships the example **`SummarizedExperiment`** used below. **DESeq2**
|
|
2107
|
+
pulls in several dependencies; the first install can take several minutes.
|
|
2108
|
+
|
|
2109
|
+
## Example: DESeq2 on the airway dataset
|
|
2110
|
+
|
|
2111
|
+
The script **`examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb`** is the canonical
|
|
2112
|
+
version in the repository. Run it from the **Galaaz repository root** with JRuby, for example:
|
|
2113
|
+
|
|
2114
|
+
```text
|
|
2115
|
+
bin/galaaz-jruby examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb
|
|
2116
|
+
```
|
|
2117
|
+
|
|
2118
|
+
The workflow in Ruby mirrors a standard DESeq2 vignette:
|
|
2119
|
+
|
|
2120
|
+
1. **`R.library('DESeq2')`** and **`R.library('airway')`**, then **`R.data('airway')`** so the
|
|
2121
|
+
object exists in R’s global environment.
|
|
2122
|
+
2. **`airway = ~R[:airway]`** pulls the experiment into a Galaaz wrapper so you can pass it to R
|
|
2123
|
+
functions as a Ruby value.
|
|
2124
|
+
3. **`R.DESeqDataSet(..., design: (R[:all].til R[:cell] + R[:dex]))`** builds the **`DESeqDataSet`**. The
|
|
2125
|
+
**`(R[:all].til R[:cell] + R[:dex])`** form is Galaaz’s way of passing the one-sided formula
|
|
2126
|
+
**`~ cell + dex`** (adjust for the design you need).
|
|
2127
|
+
4. Prefilter rows with almost no counts: **`keep = R.rowSums(R.counts(dds)) >= 10`** and
|
|
2128
|
+
**`dds = dds[keep, :all]`**.
|
|
2129
|
+
5. **`dds = R.DESeq(dds)`** fits the model; **`res = R.results(dds, contrast: R.c('dex', 'trt', 'untrt'))`**
|
|
2130
|
+
extracts the treatment contrast (adjust **`contrast`** for your experiment).
|
|
2131
|
+
6. Summaries use normal Ruby string interpolation on **`R.nrow`**, **`R.ncol`**, **`R.colnames`**, etc.
|
|
2132
|
+
7. **`R.pdf(...); R.plotMA(res, ...); R.dev__off`** writes DESeq2’s MA plot (path is relative to the
|
|
2133
|
+
process working directory—use the repo root when running the bundled script).
|
|
2134
|
+
|
|
2135
|
+
Related benchmarks and warm-run notes live under **`docs/deseq2_airway_benchmark.md`** and
|
|
2136
|
+
**`examples/bioconductor_deseq2_airway/bench_*.rb`**.
|
|
2137
|
+
|
|
2138
|
+
Below is the full listing (same as the file in the repository). It is **not** executed while this
|
|
2139
|
+
manual is knitted, because **DESeq2** is heavy and may be absent on the build machine.
|
|
2140
|
+
|
|
2141
|
+
```{ruby deseq2_airway_full_listing, eval=FALSE}
|
|
2142
|
+
# Canonical script: examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb
|
|
2143
|
+
# Run: bin/galaaz-jruby examples/.../deseq2_airway_galaaz.rb (repo root).
|
|
2144
|
+
|
|
2145
|
+
require 'galaaz'
|
|
2146
|
+
|
|
2147
|
+
R.library('DESeq2')
|
|
2148
|
+
R.library('airway')
|
|
2149
|
+
R.data('airway')
|
|
2150
|
+
|
|
2151
|
+
airway = ~R[:airway]
|
|
2152
|
+
|
|
2153
|
+
# Build DESeq2 dataset with one-sided formula: ~ cell + dex.
|
|
2154
|
+
dds = R.DESeqDataSet(airway, design: (R[:all].til R[:cell] + R[:dex]))
|
|
2155
|
+
|
|
2156
|
+
# Prefilter genes with almost no counts.
|
|
2157
|
+
keep = R.rowSums(R.counts(dds)) >= 10
|
|
2158
|
+
dds = dds[keep, :all]
|
|
2159
|
+
|
|
2160
|
+
# Fit DE model and extract treatment effect.
|
|
2161
|
+
dds = R.DESeq(dds)
|
|
2162
|
+
res = R.results(dds, contrast: R.c('dex', 'trt', 'untrt'))
|
|
2163
|
+
|
|
2164
|
+
# Compact sanity outputs for quick verification.
|
|
2165
|
+
puts "Samples: #{R.ncol(dds)}"
|
|
2166
|
+
puts "Genes after prefilter: #{R.nrow(dds)}"
|
|
2167
|
+
puts "Result rows: #{R.nrow(res)}"
|
|
2168
|
+
puts "Result columns: #{R.colnames(res)}"
|
|
2169
|
+
puts "Significant genes (padj < 0.05): #{R.sum(res.padj < 0.05, na__rm: true)}"
|
|
2170
|
+
|
|
2171
|
+
res_ordered = res[R.order(res.padj), :all]
|
|
2172
|
+
puts R.head(R.as__data__frame(res_ordered), 10)
|
|
2173
|
+
|
|
2174
|
+
# Standard DESeq2 plot call written to file.
|
|
2175
|
+
R.pdf('examples/bioconductor_deseq2_airway/plotMA_galaaz.pdf')
|
|
2176
|
+
R.plotMA(res, ylim: R.c(-5, 5))
|
|
2177
|
+
R.dev__off
|
|
2178
|
+
```
|
|
2179
|
+
|
|
2180
|
+
If **DESeq2** and **airway** are installed, the next chunk loads the data and prints a short
|
|
2181
|
+
preview (it does **not** run **`DESeq`** so the manual knits quickly).
|
|
2182
|
+
|
|
2183
|
+
```{ruby deseq2_airway_smoke, message=FALSE, warning=FALSE}
|
|
2184
|
+
unless R::Support.eval("requireNamespace('DESeq2', quietly=TRUE) && requireNamespace('airway', quietly=TRUE)")
|
|
2185
|
+
puts '(Skip: install DESeq2 and airway via BiocManager in R to run the full example.)'
|
|
2186
|
+
else
|
|
2187
|
+
R.library('DESeq2')
|
|
2188
|
+
R.library('airway')
|
|
2189
|
+
R.data('airway')
|
|
2190
|
+
airway = ~R[:airway]
|
|
2191
|
+
puts 'airway object (head of assay / dims via R):'
|
|
2192
|
+
puts "ncol(samples): #{R.ncol(airway)}"
|
|
2193
|
+
puts R.head(R.assay(airway), 3)
|
|
2194
|
+
end
|
|
824
2195
|
```
|
|
825
2196
|
|
|
2197
|
+
# Performance
|
|
2198
|
+
|
|
2199
|
+
For realistic analyses, **most wall-clock time is spent inside GNU R** (model fitting, I/O inside
|
|
2200
|
+
R, graphics). The Galaaz **bridge** adds overhead mainly from **starting a session**, **serializing
|
|
2201
|
+
requests**, and **wrapping results** in Ruby objects—not from reimplementing R’s numerical work.
|
|
2202
|
+
|
|
2203
|
+
Practical tips:
|
|
2204
|
+
|
|
2205
|
+
* Keep **hot loops** in R or vectorized code when possible; use Ruby for orchestration, I/O, and
|
|
2206
|
+
glue.
|
|
2207
|
+
* **Reuse one process**: running many short scripts cold-starts Ruby, the JVM, and R each time;
|
|
2208
|
+
a long-lived process or repeated calls in one run amortize setup (see benchmarks below).
|
|
2209
|
+
* **Batch data**: merge shards in Ruby, then call **`R::Arrow.from_ruby_batches`** (or build one
|
|
2210
|
+
data frame) instead of millions of tiny R calls.
|
|
2211
|
+
|
|
2212
|
+
For measured discussion (including DESeq2-style workloads and warm comparisons), see
|
|
2213
|
+
**`docs/performance.md`** and **`docs/deseq2_airway_benchmark.md`** in the Galaaz repository.
|
|
2214
|
+
|
|
826
2215
|
# Graphics in Galaaz
|
|
827
2216
|
|
|
828
2217
|
Creating graphics in Galaaz is quite easy, as it can use all the power of ggplot2. There are
|
|
829
|
-
many resources
|
|
2218
|
+
many resources on the web that teach ggplot, so here we give a quick example of ggplot
|
|
830
2219
|
integration with Ruby. We continue to use the :mtcars dataset and we will plot a diverging
|
|
831
|
-
bar plot, showing cars that have 'above' or 'below' gas
|
|
2220
|
+
bar plot, showing cars that have 'above' or 'below' gas consumption. Let's first prepare
|
|
832
2221
|
the data frame with the necessary data:
|
|
833
2222
|
|
|
834
2223
|
```{ruby diverging_plot_pre}
|
|
835
|
-
#
|
|
836
|
-
mtcars =
|
|
837
|
-
|
|
838
|
-
#
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
mtcars.
|
|
842
|
-
|
|
843
|
-
# compute normalized mpg and add it to a new column called mpg_z
|
|
844
|
-
# Note that the mean value for mpg can be obtained by calling the 'mean'
|
|
845
|
-
# function on the vector 'mtcars.mpg'. The same with the standard
|
|
846
|
-
# deviation 'sd'. The vector is then rounded to two digits with 'round 2'
|
|
2224
|
+
# :mtcars -> Ruby handle
|
|
2225
|
+
mtcars = ~R[:mtcars]
|
|
2226
|
+
|
|
2227
|
+
# Row labels are not a plot column; copy them to car_name.
|
|
2228
|
+
mtcars.car_name = R.rownames(R[:mtcars])
|
|
2229
|
+
|
|
2230
|
+
# Z-score mpg (mean/sd on mtcars.mpg); round to 2 decimals.
|
|
847
2231
|
mtcars.mpg_z = ((mtcars.mpg - mtcars.mpg.mean)/mtcars.mpg.sd).round 2
|
|
848
2232
|
|
|
849
|
-
#
|
|
850
|
-
# that looks at every element of the mpg_z vector and if the value is below
|
|
851
|
-
# 0, returns 'below', otherwise returns 'above'
|
|
2233
|
+
# ifelse is vectorized: below / above average mpg_z.
|
|
852
2234
|
mtcars.mpg_type = (mtcars.mpg_z < 0).ifelse("below", "above")
|
|
853
2235
|
|
|
854
|
-
#
|
|
2236
|
+
# Sort rows by mpg_z.
|
|
855
2237
|
mtcars = mtcars[mtcars.mpg_z.order, :all]
|
|
856
2238
|
|
|
857
|
-
#
|
|
2239
|
+
# Factor car_name so plot order follows sort.
|
|
858
2240
|
mtcars.car_name = mtcars.car_name.factor levels: mtcars.car_name
|
|
859
2241
|
|
|
860
|
-
# let's look at the final data frame
|
|
861
2242
|
puts mtcars.head
|
|
862
2243
|
```
|
|
863
|
-
Now,
|
|
864
|
-
|
|
2244
|
+
Now, let's plot the diverging bar plot. When using gKnit, you normally do **not** need to open a
|
|
2245
|
+
graphics device manually; gKnit arranges the figure device for chunk output. Galaaz
|
|
865
2246
|
provides integration with ggplot. The interested reader should check online for more
|
|
866
2247
|
information on ggplot, since it is outside the scope of this manual describing
|
|
867
|
-
how ggplot works.
|
|
2248
|
+
how ggplot works. Here we give only a brief description of how this plot is generated.
|
|
868
2249
|
|
|
869
|
-
ggplot implements the 'grammar of graphics'. In this approach, plots are
|
|
2250
|
+
ggplot implements the 'grammar of graphics'. In this approach, plots are built by
|
|
870
2251
|
adding layers to the plot. On the first layer we describe what we want on the 'x'
|
|
871
2252
|
and 'y' axis of the plot. In this case, we have 'car_name' on the 'x' axis and
|
|
872
2253
|
'mpg\_z' on the 'y' axis. Then the type of graph is specified by adding
|
|
873
2254
|
'geom\_bar' (for a bar graph). We specify that our bars should be filled using
|
|
874
|
-
'mpg\_type', which is either 'above' or '
|
|
2255
|
+
'mpg\_type', which is either 'above' or 'below' giving then two colours for
|
|
875
2256
|
filling. On the next layer we specify the labels for the graph, then we add the
|
|
876
2257
|
title and subtitle. Finally, in a bar chart usually bars go on the vertical direction,
|
|
877
|
-
but in this graph we want the bars to be horizontally
|
|
2258
|
+
but in this graph we want the bars to be horizontally laid so we add 'coord\_flip'.
|
|
878
2259
|
|
|
879
2260
|
```{ruby diverging_bar, fig.width = 9.1, fig.height = 6.5}
|
|
880
2261
|
require 'ggplot'
|
|
@@ -892,7 +2273,7 @@ puts mtcars.ggplot(E.aes(x: :car_name, y: :mpg_z, label: :mpg_z)) +
|
|
|
892
2273
|
# Coding with Tidyverse
|
|
893
2274
|
|
|
894
2275
|
In R, and when coding with 'tidyverse', arguments to a function are usually not
|
|
895
|
-
*
|
|
2276
|
+
*referentially transparent*. That is, you can’t replace a value with a seemingly equivalent
|
|
896
2277
|
object that you’ve defined elsewhere. To see the problem, let's first define a data frame:
|
|
897
2278
|
|
|
898
2279
|
```{ruby df}
|
|
@@ -908,17 +2289,17 @@ filter(df, my_var == 1)
|
|
|
908
2289
|
```
|
|
909
2290
|
It generates the following error: "object 'x' not found.
|
|
910
2291
|
|
|
911
|
-
However, in Galaaz, arguments are
|
|
912
|
-
code
|
|
2292
|
+
However, in Galaaz, arguments are referentially transparent as can be seen by the
|
|
2293
|
+
code below. Note initially that 'my_var = R[:x]' will not give the error "object 'x' not found"
|
|
913
2294
|
since ':x' is treated as an expression and assigned to my\_var. Then when doing (my\_var.eq 1),
|
|
914
|
-
my\_var is a variable that resolves to ':x' and it becomes equivalent to (:x.eq 1) which is
|
|
2295
|
+
my\_var is a variable that resolves to ':x' and it becomes equivalent to (R[:x].eq 1) which is
|
|
915
2296
|
what we want.
|
|
916
2297
|
|
|
917
2298
|
```{ruby my_var}
|
|
918
|
-
my_var = :x
|
|
2299
|
+
my_var = R[:x]
|
|
919
2300
|
puts df.filter(my_var.eq 1)
|
|
920
2301
|
```
|
|
921
|
-
As stated by
|
|
2302
|
+
As stated by Hadley
|
|
922
2303
|
|
|
923
2304
|
> dplyr code is ambiguous. Depending on what variables are defined where,
|
|
924
2305
|
> filter(df, x == y) could be equivalent to any of:
|
|
@@ -930,9 +2311,9 @@ df[x == df$y, ]
|
|
|
930
2311
|
df[x == y, ]
|
|
931
2312
|
```
|
|
932
2313
|
In galaaz this ambiguity does not exist, filter(df, x.eq y) is not a valid expression as
|
|
933
|
-
expressions are build with symbols. In doing filter(df, :x.eq y) we are looking for elements
|
|
2314
|
+
expressions are build with symbols. In doing filter(df, R[:x].eq y) we are looking for elements
|
|
934
2315
|
of the 'x' column that are equal to a previously defined y variable. Finally in
|
|
935
|
-
filter(df, :x.eq :y) we are looking for elements in which the 'x' column value is equal to
|
|
2316
|
+
filter(df, R[:x].eq R[:y]) we are looking for elements in which the 'x' column value is equal to
|
|
936
2317
|
the 'y' column value. This can be seen in the following two chunks of code:
|
|
937
2318
|
|
|
938
2319
|
```{ruby disamb1}
|
|
@@ -940,13 +2321,13 @@ y = 1
|
|
|
940
2321
|
x = 2
|
|
941
2322
|
|
|
942
2323
|
# looking for values where the 'x' column is equal to the 'y' column
|
|
943
|
-
puts df.filter(:x.eq :y)
|
|
2324
|
+
puts df.filter(R[:x].eq R[:y])
|
|
944
2325
|
```
|
|
945
2326
|
|
|
946
2327
|
```{ruby disamb2}
|
|
947
2328
|
# looking for values where the 'x' column is equal to the 'y' variable
|
|
948
2329
|
# in this case, the number 1
|
|
949
|
-
puts df.filter(:x.eq y)
|
|
2330
|
+
puts df.filter(R[:x].eq y)
|
|
950
2331
|
```
|
|
951
2332
|
## Writing a function that applies to different data sets
|
|
952
2333
|
|
|
@@ -973,11 +2354,12 @@ Unfortunately, in R, this function can fail silently if one of the variables isn
|
|
|
973
2354
|
in the data frame, but is present in the global environment. We will not go through here how
|
|
974
2355
|
to solve this problem in R.
|
|
975
2356
|
|
|
976
|
-
In Galaaz the method mutate_y
|
|
2357
|
+
In Galaaz the method mutate_y below will work fine and will never fail silently.
|
|
977
2358
|
|
|
978
2359
|
```{ruby mutate_y, warning=FALSE}
|
|
979
2360
|
def mutate_y(df)
|
|
980
|
-
|
|
2361
|
+
# Mutate column names are Ruby kwargs (y: …). Use .assign only for R `<-` expressions.
|
|
2362
|
+
df.mutate(y: R[:a] + R[:x])
|
|
981
2363
|
end
|
|
982
2364
|
```
|
|
983
2365
|
Here we create a data frame that has only one column named 'x':
|
|
@@ -987,8 +2369,8 @@ df1 = R.data__frame(x: (1..3))
|
|
|
987
2369
|
puts df1
|
|
988
2370
|
```
|
|
989
2371
|
|
|
990
|
-
Note that method mutate_y will fail
|
|
991
|
-
in the scope of the method. Variable 'a' has no relationship with the symbol
|
|
2372
|
+
Note that method mutate_y will fail independently from the fact that variable 'a' is defined and
|
|
2373
|
+
in the scope of the method. Variable 'a' has no relationship with the symbol `R[:a]` used in the
|
|
992
2374
|
definition of 'mutate\_y' above:
|
|
993
2375
|
|
|
994
2376
|
```{ruby call_mutate_y, warning = FALSE}
|
|
@@ -997,12 +2379,13 @@ mutate_y(df1)
|
|
|
997
2379
|
```
|
|
998
2380
|
## Different expressions
|
|
999
2381
|
|
|
1000
|
-
Let's move to the next problem as presented by
|
|
2382
|
+
Let's move to the next problem as presented by Hadley where trying to write a function in R
|
|
1001
2383
|
that will receive two argumens, the first a variable and the second an expression is not trivial.
|
|
1002
|
-
|
|
2384
|
+
Below we create a data frame and we want to write a function that groups data by a variable and
|
|
1003
2385
|
summarises it by an expression:
|
|
1004
2386
|
|
|
1005
2387
|
```{r diff_expr}
|
|
2388
|
+
library(dplyr)
|
|
1006
2389
|
set.seed(123)
|
|
1007
2390
|
|
|
1008
2391
|
df <- data.frame(
|
|
@@ -1027,7 +2410,7 @@ d2 <- df %>%
|
|
|
1027
2410
|
as.data.frame(d2)
|
|
1028
2411
|
```
|
|
1029
2412
|
|
|
1030
|
-
As shown by
|
|
2413
|
+
As shown by Hadley, one might expect this function to do the trick:
|
|
1031
2414
|
|
|
1032
2415
|
```{r diff_exp_fnc}
|
|
1033
2416
|
my_summarise <- function(df, group_var) {
|
|
@@ -1042,32 +2425,32 @@ my_summarise <- function(df, group_var) {
|
|
|
1042
2425
|
|
|
1043
2426
|
In order to solve this problem, coding with dplyr requires the introduction of many new concepts
|
|
1044
2427
|
and functions such as 'quo', 'quos', 'enquo', 'enquos', '!!' (bang bang), '!!!' (triple bang).
|
|
1045
|
-
Again, we'll leave to
|
|
2428
|
+
Again, we'll leave to Hadley the explanation on how to use all those functions.
|
|
1046
2429
|
|
|
1047
2430
|
Now, let's try to implement the same function in galaaz. The next code block first prints the
|
|
1048
|
-
'df' data frame defined previously in R (to access an R variable from Galaaz, we use the
|
|
1049
|
-
operator
|
|
2431
|
+
'df' data frame defined previously in R (to access an R variable from Galaaz, we use the tilde
|
|
2432
|
+
operator `~` applied to the R variable name as a symbol, e.g. `:df`).
|
|
1050
2433
|
|
|
1051
2434
|
```{ruby r_dataframe}
|
|
1052
|
-
puts
|
|
2435
|
+
puts ~R[:df]
|
|
1053
2436
|
```
|
|
1054
2437
|
|
|
1055
2438
|
We then create the 'my_summarize' method and call it passing the R data frame and
|
|
1056
|
-
the group by variable ':g1':
|
|
2439
|
+
the group by variable 'R[:g1]':
|
|
1057
2440
|
|
|
1058
2441
|
```{ruby diff_exp_ruby_func}
|
|
1059
2442
|
def my_summarize(df, group_var)
|
|
1060
2443
|
df.group_by(group_var).
|
|
1061
|
-
summarize(a: :a.mean)
|
|
2444
|
+
summarize(a: R[:a].mean)
|
|
1062
2445
|
end
|
|
1063
2446
|
|
|
1064
|
-
puts my_summarize(:df, :g1)
|
|
2447
|
+
puts my_summarize(~R[:df], R[:g1])
|
|
1065
2448
|
```
|
|
1066
2449
|
|
|
1067
2450
|
It works!!! Well, let's make sure this was not just some coincidence
|
|
1068
2451
|
|
|
1069
2452
|
```{ruby group_g2}
|
|
1070
|
-
puts my_summarize(:df, :g2)
|
|
2453
|
+
puts my_summarize(~R[:df], R[:g2])
|
|
1071
2454
|
```
|
|
1072
2455
|
|
|
1073
2456
|
Great, everything is fine! No magic, no new functions, no complexities, just normal, standard Ruby
|
|
@@ -1079,7 +2462,7 @@ In the previous section we've managed to get rid of all NSE formulation for a si
|
|
|
1079
2462
|
does this remain true for more complex examples, or will the Galaaz way prove inpractical for
|
|
1080
2463
|
more complex code?
|
|
1081
2464
|
|
|
1082
|
-
In the next example
|
|
2465
|
+
In the next example Hadley proposes us to write a function that given an expression such as 'a'
|
|
1083
2466
|
or 'a * b', calculates three summaries. What we want a function that does the same as these R
|
|
1084
2467
|
statements:
|
|
1085
2468
|
|
|
@@ -1108,9 +2491,9 @@ def my_summarise2(df, expr)
|
|
|
1108
2491
|
)
|
|
1109
2492
|
end
|
|
1110
2493
|
|
|
1111
|
-
puts my_summarise2((
|
|
2494
|
+
puts my_summarise2((~R[:df]), :a)
|
|
1112
2495
|
puts "\n"
|
|
1113
|
-
puts my_summarise2((
|
|
2496
|
+
puts my_summarise2((~R[:df]), R[:a] * R[:b])
|
|
1114
2497
|
```
|
|
1115
2498
|
|
|
1116
2499
|
Once again, there is no need to use any special theory or functions. The only point to be
|
|
@@ -1118,7 +2501,7 @@ careful about is the use of 'E' to build expressions from functions 'mean', 'sum
|
|
|
1118
2501
|
|
|
1119
2502
|
## Different input and output variable
|
|
1120
2503
|
|
|
1121
|
-
Now the next challenge presented by
|
|
2504
|
+
Now the next challenge presented by Hadley is to vary the name of the output variables based on
|
|
1122
2505
|
the received expression. So, if the input expression is 'a', we want our data frame columns to
|
|
1123
2506
|
be named 'mean\_a' and 'sum\_a'. Now, if the input expression is 'b', columns
|
|
1124
2507
|
should be named 'mean\_b' and 'sum\_b'.
|
|
@@ -1144,7 +2527,7 @@ mutate(df, mean_b = mean(b), sum_b = sum(b))
|
|
|
1144
2527
|
#> 4 2 2 5 4 3 15
|
|
1145
2528
|
#> # … with 1 more row
|
|
1146
2529
|
```
|
|
1147
|
-
In order to solve this problem in R,
|
|
2530
|
+
In order to solve this problem in R, Hadley needs to introduce some more new functions and notations:
|
|
1148
2531
|
'quo_name' and the ':=' operator from package 'rlang'
|
|
1149
2532
|
|
|
1150
2533
|
Here is our Ruby code:
|
|
@@ -1158,9 +2541,9 @@ def my_mutate(df, expr)
|
|
|
1158
2541
|
sum_name => E.sum(expr))
|
|
1159
2542
|
end
|
|
1160
2543
|
|
|
1161
|
-
puts my_mutate((
|
|
2544
|
+
puts my_mutate((~R[:df]), :a)
|
|
1162
2545
|
puts "\n"
|
|
1163
|
-
puts my_mutate((
|
|
2546
|
+
puts my_mutate((~R[:df]), :b)
|
|
1164
2547
|
```
|
|
1165
2548
|
It really seems that "Non Standard Evaluation" is actually quite standard in Galaaz! But, you
|
|
1166
2549
|
might have noticed a small change in the way the arguments to the mutate method were called.
|
|
@@ -1172,7 +2555,7 @@ and variable mean\_name is not followed by ':' but by '=>'. This is standard Ru
|
|
|
1172
2555
|
|
|
1173
2556
|
## Capturing multiple variables
|
|
1174
2557
|
|
|
1175
|
-
Moving on with new complexities,
|
|
2558
|
+
Moving on with new complexities, Hadley proposes us to solve the problem in which the
|
|
1176
2559
|
summarise function will receive any number of grouping variables.
|
|
1177
2560
|
|
|
1178
2561
|
This again is quite standard Ruby. In order to receive an undefined number of paramenters
|
|
@@ -1184,7 +2567,7 @@ def my_summarise3(df, *group_vars)
|
|
|
1184
2567
|
summarise(a: E.mean(:a))
|
|
1185
2568
|
end
|
|
1186
2569
|
|
|
1187
|
-
puts my_summarise3((
|
|
2570
|
+
puts my_summarise3((~R[:df]), R[:g1], R[:g2])
|
|
1188
2571
|
```
|
|
1189
2572
|
|
|
1190
2573
|
## Why does R require NSE and Galaaz does not?
|
|
@@ -1202,7 +2585,7 @@ In Ruby, there is no lazy evaluation of parameters and 'a' is always a variable
|
|
|
1202
2585
|
Variables assume their value as soon as they are used, so 'x = a' is immediately evaluate and
|
|
1203
2586
|
variable 'x' will receive the value of variable 'a' as soon as the Ruby statement is executed.
|
|
1204
2587
|
Ruby also provides the notion of a symbol; ':a' is a symbol and does not evaluate to anything.
|
|
1205
|
-
Galaaz uses Ruby symbols to build expressions that are not bound to anything: ':a.eq :b' is
|
|
2588
|
+
Galaaz uses Ruby symbols to build expressions that are not bound to anything: 'R[:a].eq R[:b]' is
|
|
1206
2589
|
clearly an expression and has no relationship whatsoever with the statment 'a = b'. By using
|
|
1207
2590
|
symbols, variables and expressions all the possible ambiguities that are found in R are
|
|
1208
2591
|
eliminated in Galaaz.
|
|
@@ -1212,7 +2595,7 @@ of input they are expecting, they might be expecting regular variables or they m
|
|
|
1212
2595
|
expecting expressions and the R function will know how to deal with an input of the form
|
|
1213
2596
|
'a = b', now for the Ruby developer it might not be immediately clear if it should call the
|
|
1214
2597
|
function passing the value 'true' if variable 'a' is equal to variable 'b' or if it should
|
|
1215
|
-
call the function passing the expression ':a.eq :b'.
|
|
2598
|
+
call the function passing the expression 'R[:a].eq R[:b]'.
|
|
1216
2599
|
|
|
1217
2600
|
|
|
1218
2601
|
## Advanced dplyr features
|
|
@@ -1235,12 +2618,13 @@ In the following examples, we show the use of functions 'group\_by\_at', 'summar
|
|
|
1235
2618
|
features of characters in the Starwars movies:
|
|
1236
2619
|
|
|
1237
2620
|
```{ruby starwars}
|
|
1238
|
-
puts (
|
|
2621
|
+
puts (~R[:starwars]).head
|
|
1239
2622
|
```
|
|
1240
|
-
The grouped_mean function
|
|
2623
|
+
The grouped_mean function below will receive a grouping variable and calculate summaries for
|
|
1241
2624
|
the value\_variables given:
|
|
1242
2625
|
|
|
1243
2626
|
```{r grouped_mean}
|
|
2627
|
+
library(dplyr)
|
|
1244
2628
|
grouped_mean <- function(data, grouping_variables, value_variables) {
|
|
1245
2629
|
data %>%
|
|
1246
2630
|
group_by_at(grouping_variables) %>%
|
|
@@ -1262,24 +2646,26 @@ def grouped_mean(data, grouping_variables, value_variables)
|
|
|
1262
2646
|
data.
|
|
1263
2647
|
group_by_at(grouping_variables).
|
|
1264
2648
|
mutate(count: E.n).
|
|
1265
|
-
summarise_at(E.c(value_variables, "count"),
|
|
2649
|
+
summarise_at(E.c(value_variables, "count"), ~R[:mean], na__rm: true).
|
|
1266
2650
|
rename_at(value_variables, E.funs(E.paste0("mean_", value_variables)))
|
|
1267
2651
|
end
|
|
1268
2652
|
|
|
1269
|
-
puts grouped_mean((
|
|
2653
|
+
puts grouped_mean((~R[:starwars]), "eye_color", E.c("mass", "birth_year"))
|
|
1270
2654
|
```
|
|
1271
2655
|
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
2656
|
+
The examples above cover programmatic dplyr with string column names and `_at` helpers. The same
|
|
2657
|
+
Galaaz patterns (symbols, `E.*` for expression-safe functions, and Ruby methods on R-backed objects)
|
|
2658
|
+
extend to other tidyverse workflows; consult R package documentation for function-specific
|
|
2659
|
+
arguments.
|
|
1275
2660
|
|
|
1276
2661
|
# Contributing
|
|
1277
2662
|
|
|
1278
|
-
|
|
1279
2663
|
* Fork it
|
|
1280
|
-
* Create your feature branch (git checkout -b my-new-feature)
|
|
1281
|
-
* Write
|
|
1282
|
-
|
|
1283
|
-
*
|
|
1284
|
-
*
|
|
1285
|
-
|
|
2664
|
+
* Create your feature branch (`git checkout -b my-new-feature`)
|
|
2665
|
+
* Write tests — use **`bin/run_rspec`** or **`bin/run_all_rspec`** with **JRuby** so JVM flags and
|
|
2666
|
+
the load path match **`docs/testing.md`**
|
|
2667
|
+
* Commit your changes (`git commit -am 'Add some feature'`)
|
|
2668
|
+
* Push to the branch (`git push origin my-new-feature`)
|
|
2669
|
+
* Open a pull request
|
|
2670
|
+
|
|
2671
|
+
# References
|