galaaz 0.5.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +26 -0
- data/LICENSE +0 -0
- data/README.md +1360 -636
- data/Rakefile +61 -41
- data/bin/galaaz-bootstrap +137 -0
- data/bin/galaaz-jruby +14 -0
- data/bin/galaaz_jruby_env.inc.sh +6 -0
- data/bin/gbookdown +64 -0
- data/bin/gknit +84 -13
- data/bin/gknit-draft.rb +0 -0
- data/bin/gstudio +5 -3
- data/bin/gstudio_irb.rb +0 -0
- data/bin/gstudio_pry.rb +0 -0
- data/bin/install-tinytex +6 -0
- data/bin/run_all_rspec +43 -0
- data/bin/run_example +14 -0
- data/bin/run_old_rspec +19 -0
- data/bin/run_rspec +23 -0
- data/bin/run_rspec_subset +38 -0
- data/bin/run_slow_rspec +19 -0
- data/blogs/R-on-Rails-Planning-Document.md +940 -0
- data/blogs/README.md +100 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot.Rmd +38 -66
- data/blogs/galaaz_ggplot/galaaz_ggplot.log +754 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot.md +115 -155
- data/blogs/galaaz_ggplot/galaaz_ggplot.tex +607 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-html/midwest_rb.png +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-html/scatter_plot_rb.png +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-markdown_github/midwest_rb.png +0 -0
- data/blogs/galaaz_ggplot/galaaz_ggplot_files/figure-markdown_github/scatter_plot_rb.png +0 -0
- data/blogs/galaaz_ggplot/midwest.Rmd +3 -3
- data/blogs/galaaz_ggplot/midwest_external_png +0 -0
- data/blogs/gknit/gknit.Rmd +47 -52
- data/blogs/gknit/gknit.md +1430 -0
- data/blogs/gknit/gknit_files/figure-html/bubble-1.png +0 -0
- data/blogs/gknit/gknit_files/figure-html/diverging_bar.png +0 -0
- data/blogs/gknit/lst.rds +0 -0
- data/blogs/gknit/model.rb +1 -1
- data/blogs/gknit/stats.bib +0 -0
- data/blogs/manual/include_model_local_repro.Rmd +14 -0
- data/blogs/manual/include_model_local_repro.md +75 -0
- data/blogs/manual/lst.rds +0 -0
- data/blogs/manual/manual.Rmd +852 -239
- data/blogs/manual/manual.log +1786 -0
- data/blogs/manual/manual.md +1360 -636
- data/blogs/manual/manual.tex +1883 -1161
- data/blogs/manual/manual_files/figure-html/bubble-1.png +0 -0
- data/blogs/manual/manual_files/figure-html/diverging_bar.png +0 -0
- data/blogs/manual/manual_files/figure-latex/bubble-1.png +0 -0
- data/blogs/manual/model.rb +1 -1
- data/blogs/nse_dplyr/nse_dplyr.Rmd +84 -111
- data/blogs/nse_dplyr/nse_dplyr.log +928 -0
- data/blogs/nse_dplyr/nse_dplyr.md +198 -229
- data/blogs/oh_my/not_so.rb +0 -0
- data/blogs/oh_my/oh_my.Rmd +1234 -25
- data/blogs/oh_my/oh_my.log +804 -0
- data/blogs/oh_my/oh_my.md +1663 -86
- data/blogs/oh_my/oh_my.tex +821 -0
- data/blogs/oh_my/old.Rmd +15 -14
- data/blogs/ruby_plot/ruby_plot.Rmd +58 -82
- data/blogs/ruby_plot/ruby_plot.log +885 -0
- data/blogs/ruby_plot/ruby_plot.md +71 -102
- data/blogs/ruby_plot/ruby_plot.tex +940 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/dose_len.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_delivery.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facet_by_dose.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_by_delivery_color2.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_decorations.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/facets_with_points.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/final_box_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/final_violin_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-html/violin_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/dose_len.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facet_by_delivery.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facet_by_dose.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_by_delivery_color.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_by_delivery_color2.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_decorations.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_jitter.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/facets_with_points.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/final_box_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/final_violin_plot.png +0 -0
- data/blogs/ruby_plot/ruby_plot_files/figure-latex/violin_with_jitter.png +0 -0
- data/blogs/test/test.Rmd +14 -0
- data/examples/50Plots_MasterList/Images/midwest-scatterplot.PNG +0 -0
- data/examples/50Plots_MasterList/ScatterPlot.rb +0 -0
- data/examples/50Plots_MasterList/scatter_plot.rb +0 -0
- data/examples/Bibliography/master.bib +0 -0
- data/examples/Bibliography/stats.bib +0 -0
- data/examples/R/calc.R +0 -0
- data/examples/R/java_interop.R +0 -0
- data/examples/bioconductor_deseq2_airway/Documentation/DESeq2-airway-walkthrough.md +56 -0
- data/examples/bioconductor_deseq2_airway/bench_galaaz_three_same_process.rb +53 -0
- data/examples/bioconductor_deseq2_airway/bench_r_three_same_process.R +34 -0
- data/examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb +33 -0
- data/examples/bioconductor_deseq2_airway/deseq2_airway_galaaz_optimized.rb +34 -0
- data/examples/bioconductor_deseq2_airway/deseq2_airway_minimal.R +30 -0
- data/examples/bioconductor_deseq2_airway/deseq2_airway_pipeline_for_bench.R +36 -0
- data/examples/islr/all.rb +13 -0
- data/examples/islr/ch2.spec.rb +37 -7
- data/examples/islr/ch3.spec.rb +11 -2
- data/examples/islr/ch3_boston.rb +27 -0
- data/examples/islr/ch3_multiple_regression.rb +0 -0
- data/examples/islr/ch6.spec.rb +24 -1
- data/examples/islr/x_y_rnorm.jpg +0 -0
- data/examples/latex_templates/Test-acm_article/acm_proc_article-sp.cls +0 -0
- data/examples/latex_templates/Test-acm_article/sigproc.bib +0 -0
- data/examples/latex_templates/Test-acs_article/acs-Test-acs_article.bib +0 -0
- data/examples/latex_templates/Test-acs_article/acs-my_output.bib +0 -0
- data/examples/latex_templates/Test-aea_article/BibFile.bib +0 -0
- data/examples/latex_templates/Test-aea_article/Test-aea_article.Rmd +0 -0
- data/examples/latex_templates/Test-aea_article/references.bib +0 -0
- data/examples/latex_templates/Test-amq_article/Test-amq_article.Rmd +0 -0
- data/examples/latex_templates/Test-amq_article/Test-amq_article.pdfsync +0 -0
- data/examples/latex_templates/Test-ieee_article/IEEEtran.bst +0 -0
- data/examples/latex_templates/Test-ieee_article/mybibfile.bib +0 -0
- data/examples/latex_templates/Test-rjournal_article/RJournal.sty +0 -0
- data/examples/latex_templates/Test-rjournal_article/RJreferences.bib +0 -0
- data/examples/latex_templates/Test-rjournal_article/Test-rjournal_article.Rmd +0 -0
- data/examples/misc/baseball.csv +0 -0
- data/examples/misc/ggplot.rb +3 -2
- data/examples/misc/moneyball.rb +0 -0
- data/examples/misc/subsetting.rb +0 -0
- data/examples/multithread_shards_to_r/shards_to_r.rb +67 -0
- data/examples/rmarkdown/svm-rmarkdown-anon-ms-example/svm-rmarkdown-anon-ms-example.Rmd +0 -0
- data/examples/rmarkdown/svm-rmarkdown-article-example/svm-rmarkdown-article-example.Rmd +0 -0
- data/examples/rmarkdown/svm-rmarkdown-beamer-example/svm-rmarkdown-beamer-example.Rmd +0 -0
- data/examples/rmarkdown/svm-rmarkdown-cv/svm-rmarkdown-cv.Rmd +0 -0
- data/examples/rmarkdown/svm-rmarkdown-syllabus-example/attend-grade-relationships.csv +0 -0
- data/examples/rmarkdown/svm-rmarkdown-syllabus-example/svm-rmarkdown-syllabus-example.Rmd +0 -0
- data/examples/rmarkdown/svm-xaringan-example/svm-xaringan-example.Rmd +0 -0
- data/examples/sthda_ggplot/README.md +0 -0
- data/examples/sthda_ggplot/RUN.md +41 -0
- data/examples/sthda_ggplot/all.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/density_gg.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_area.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_density.rb +2 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_dotplot.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_freqpoly.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/geom_histogram.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/histogram_density.rb +0 -0
- data/examples/sthda_ggplot/one_variable_continuous/stat.rb +0 -0
- data/examples/sthda_ggplot/one_variable_discrete/bar.rb +0 -0
- data/examples/sthda_ggplot/qplots/box_violin_dot.rb +0 -0
- data/examples/sthda_ggplot/qplots/scatter_plots.rb +0 -0
- data/examples/sthda_ggplot/scatter_gg.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_bivariate/geom_bin2d.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_bivariate/geom_density2d.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_bivariate/geom_hex.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_cont/geom_point.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_cont/geom_smooth.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_cont/misc.rb +0 -0
- data/examples/sthda_ggplot/two_variables_cont_function/geom_area.rb +4 -3
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_bar.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_boxplot.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_dotplot.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_jitter.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_line.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_cont/geom_violin.rb +0 -0
- data/examples/sthda_ggplot/two_variables_disc_disc/geom_jitter.rb +0 -0
- data/examples/sthda_ggplot/two_variables_error/geom_crossbar.rb +0 -0
- data/ext/new_bridge/Makefile +46 -0
- data/ext/new_bridge/galaaz_gatekeeper_phase0.cpp +12 -0
- data/ext/new_bridge/galaaz_gatekeeper_phase1.cpp +1639 -0
- data/lib/R_interface/galaaz_device.R +20 -0
- data/lib/R_interface/include_engine.R +109 -0
- data/lib/R_interface/new_bridge_adapter.rb +824 -0
- data/lib/R_interface/r.rb +177 -25
- data/lib/R_interface/r_arrow.rb +113 -0
- data/lib/R_interface/r_libs.R +3 -3
- data/lib/R_interface/r_methods.rb +13 -126
- data/lib/R_interface/r_module_s.rb +0 -0
- data/lib/R_interface/rbinary_operators.rb +20 -2
- data/lib/R_interface/rclosure.rb +5 -1
- data/lib/R_interface/rdata_frame.rb +34 -70
- data/lib/R_interface/rdevice.rb +125 -0
- data/lib/R_interface/rdevices.R +0 -0
- data/lib/R_interface/renvironment.rb +10 -4
- data/lib/R_interface/rexpression.rb +5 -1
- data/lib/R_interface/rindexed_object.rb +41 -13
- data/lib/R_interface/rlanguage.rb +20 -62
- data/lib/R_interface/rlist.rb +115 -25
- data/lib/R_interface/rlogical_operators.rb +0 -0
- data/lib/R_interface/rmatrix.rb +2 -11
- data/lib/R_interface/rmd_indexed_object.rb +5 -1
- data/lib/R_interface/robject.rb +348 -290
- data/lib/R_interface/rpkg.rb +0 -0
- data/lib/R_interface/rsupport.rb +609 -328
- data/lib/R_interface/rsupport_scope.rb +2 -1
- data/lib/R_interface/rsymbol.rb +50 -0
- data/lib/R_interface/ruby_callback.rb +2 -3
- data/lib/R_interface/ruby_extensions.rb +225 -175
- data/lib/R_interface/runary_operators.rb +0 -0
- data/lib/R_interface/rvector.rb +147 -31
- data/lib/galaaz.rb +0 -0
- data/lib/galaaz_jruby.rb +22 -0
- data/lib/gknit/diagnostics.rb +50 -0
- data/lib/gknit/draft.rb +23 -17
- data/lib/gknit/include_engine.rb +15 -7
- data/lib/gknit/knitr_engine.rb +223 -74
- data/lib/gknit/rb_engine.rb +3 -3
- data/lib/gknit/ruby_engine.rb +0 -0
- data/lib/gknit.rb +1 -0
- data/lib/new_bridge/bootstrap/windows_bootstrap.rb +285 -0
- data/lib/new_bridge/envelope.rb +51 -0
- data/lib/new_bridge/eval_result.rb +26 -0
- data/lib/new_bridge/framing.rb +39 -0
- data/lib/new_bridge/instance_pool_client.rb +38 -0
- data/lib/new_bridge/r_instance_manager.rb +404 -0
- data/lib/new_bridge/session_client.rb +530 -0
- data/lib/new_bridge/tcp_framed.rb +44 -0
- data/lib/new_bridge.rb +9 -0
- data/lib/util/exec_ruby.rb +95 -20
- data/lib/util/inline_file.rb +35 -30
- data/new_bridge_specs/benchmark_phase5_5_unboxing_spec.rb +96 -0
- data/new_bridge_specs/eval_r_async_spec.rb +113 -0
- data/new_bridge_specs/integration_phase5_1_concurrent_spec.rb +50 -0
- data/new_bridge_specs/integration_phase5_1_eval_spec.rb +16 -0
- data/new_bridge_specs/integration_phase5_1_r_api_spec.rb +25 -0
- data/new_bridge_specs/integration_phase5_1_smoke_spec.rb +31 -0
- data/new_bridge_specs/integration_phase5_2_dataframe_unboxing_spec.rb +19 -0
- data/new_bridge_specs/integration_phase5_2_handle_eval_unboxing_spec.rb +25 -0
- data/new_bridge_specs/integration_phase5_3_callback_args_spec.rb +28 -0
- data/new_bridge_specs/integration_phase5_3_callback_error_spec.rb +22 -0
- data/new_bridge_specs/integration_phase5_3_callback_timeout_spec.rb +28 -0
- data/new_bridge_specs/integration_phase5_3_callbacks_smoke_spec.rb +22 -0
- data/new_bridge_specs/integration_phase5_3_edge_cases_spec.rb +52 -0
- data/new_bridge_specs/integration_phase5_3_nested_spec.rb +30 -0
- data/new_bridge_specs/integration_phase5_4_concurrent_sessions_spec.rb +53 -0
- data/new_bridge_specs/integration_phase5_4_nested_session_callbacks_spec.rb +49 -0
- data/new_bridge_specs/integration_phase5_4_session_routing_spec.rb +38 -0
- data/new_bridge_specs/integration_phase5_5_stress_concurrency_spec.rb +52 -0
- data/new_bridge_specs/integration_phase5_5_unbox_walk_spec.rb +46 -0
- data/new_bridge_specs/phase0_protocol_spec.rb +96 -0
- data/new_bridge_specs/phase1_req_ret_spec.rb +66 -0
- data/new_bridge_specs/phase2_multi_instance_spec.rb +67 -0
- data/new_bridge_specs/phase3_callbacks_spec.rb +71 -0
- data/new_bridge_specs/phase4_2_hardening_spec.rb +252 -0
- data/new_bridge_specs/phase4_3_r_instance_manager_spec.rb +85 -0
- data/new_bridge_specs/phase4_nested_callbacks_spec.rb +123 -0
- data/r_requires/ggplot.rb +0 -0
- data/r_requires/knitr.rb +0 -0
- data/specs/all.rb +15 -11
- data/specs/arrow_from_ruby_batches_spec.rb +50 -0
- data/specs/arrow_semantics_spec.rb +64 -0
- data/specs/bridge_concurrent_spec.rb +46 -0
- data/specs/bridge_nested_spec.rb +25 -0
- data/specs/dataframe_semantics_spec.rb +122 -0
- data/specs/dataframe_single_index_logical_filter_spec.rb +21 -0
- data/specs/dispatch_probe_cache_spec.rb +38 -0
- data/specs/dispatch_probe_error_class_fallback_spec.rb +20 -0
- data/specs/dispatch_probe_fallback_spec.rb +18 -0
- data/specs/environment_semantics_spec.rb +89 -0
- data/specs/field_access_spec.rb +31 -0
- data/specs/figures/bg.jpeg +0 -0
- data/specs/figures/bg.png +0 -0
- data/specs/figures/bg.svg +168 -57
- data/specs/figures/dose_len.png +0 -0
- data/specs/figures/no_args.jpeg +0 -0
- data/specs/figures/no_args.png +0 -0
- data/specs/figures/no_args.svg +168 -57
- data/specs/figures/width_height.jpeg +0 -0
- data/specs/figures/width_height.png +0 -0
- data/specs/figures/width_height_units1.jpeg +0 -0
- data/specs/figures/width_height_units1.png +0 -0
- data/specs/figures/width_height_units2.jpeg +0 -0
- data/specs/figures/width_height_units2.png +0 -0
- data/specs/formula_semantics_spec.rb +81 -0
- data/specs/galaaz_util_exec_ruby_spec.rb +85 -0
- data/specs/galaaz_util_inline_file_spec.rb +54 -0
- data/specs/gknit_cli_option_permutation_spec.rb +24 -0
- data/specs/gknit_include_engine_spec.rb +72 -0
- data/specs/gknit_install_timeout_report_spec.rb +69 -0
- data/specs/gknit_internal_error_report_spec.rb +57 -0
- data/specs/gknit_vector_map_output_spec.rb +59 -0
- data/specs/globalenv_guardrail_spec.rb +52 -0
- data/specs/language_expression_semantics_spec.rb +145 -0
- data/specs/list_semantics_spec.rb +111 -0
- data/specs/new_bridge_bulk_dataframe_transfer_spec.rb +44 -0
- data/specs/new_bridge_bulk_vector_transfer_spec.rb +73 -0
- data/specs/new_bridge_callback_timeout_spec.rb +69 -0
- data/specs/new_bridge_eval_r_fallback_spec.rb +55 -0
- data/specs/nil_null_spec.rb +42 -0
- data/specs/object_build_phase2_spec.rb +53 -0
- data/specs/phase1_callback_bridge_spec.rb +84 -0
- data/specs/phase2_gknit_generic_rendering_guardrail_spec.rb +46 -0
- data/specs/phase2_gknit_no_raw_code_leakage_spec.rb +43 -0
- data/specs/phase3_gknit_generic_graphics_capture_spec.rb +71 -0
- data/specs/plot_device_semantics_spec.rb +28 -0
- data/specs/plot_snapshot_semantics_spec.rb +58 -0
- data/specs/protocol_result_spec.rb +236 -0
- data/specs/r_batch_fail_fast_spec.rb +47 -0
- data/specs/r_bridge_bootstrap_spec.rb +11 -0
- data/specs/r_devices.spec.rb +1 -1
- data/specs/r_eval.spec.rb +16 -18
- data/specs/r_function.spec.rb +1 -1
- data/specs/r_instance_manager_spec.rb +285 -0
- data/specs/r_list_apply.spec.rb +15 -15
- data/specs/r_matrix.spec.rb +0 -0
- data/specs/r_nse.spec.rb +5 -5
- data/specs/r_object_send_dispatch_spec.rb +13 -0
- data/specs/r_vector_comparator_spec.rb +8 -0
- data/specs/r_vector_creation.spec.rb +0 -0
- data/specs/r_vector_functions.spec.rb +0 -0
- data/specs/r_vector_object.spec.rb +0 -0
- data/specs/r_vector_operators.spec.rb +0 -0
- data/specs/r_vector_structured_scalar_reads_spec.rb +35 -0
- data/specs/r_vector_subsetting.spec.rb +0 -0
- data/specs/range_helper_spec.rb +21 -0
- data/specs/rsupport_scope_spec.rb +28 -0
- data/specs/rsupport_var_name_thread_safety_spec.rb +24 -0
- data/specs/scalar_character_spec.rb +44 -0
- data/specs/scoped_symbol_dsl_refinement_spec.rb +40 -0
- data/specs/session_env_bridge_spec.rb +25 -0
- data/specs/simplecov_bootstrap_spec.rb +10 -0
- data/specs/spec_helper.rb +10 -0
- data/specs/tmp.rb +0 -0
- data/specs/unboxing_recursion_regression_spec.rb +30 -0
- data/specs/unboxing_spec.rb +49 -0
- data/specs/verify_callbacks.rb +42 -0
- data/sty/galaaz.sty +0 -0
- data/version.rb +1 -1
- metadata +194 -64
- data/blogs/galaaz_ggplot/galaaz_ggplot.html +0 -520
- data/blogs/galaaz_ggplot/galaaz_ggplot.pdf +0 -0
- data/blogs/galaaz_ggplot/midwest.html +0 -188
- data/blogs/gknit/gknit.html +0 -2266
- data/blogs/gknit/gknit.pdf +0 -0
- data/blogs/manual/manual.html +0 -4638
- data/blogs/manual/manual.pdf +0 -0
- data/blogs/manual/manual_files/figure-latex/diverging_bar.pdf +0 -0
- data/blogs/nse_dplyr/nse_dplyr.html +0 -878
- data/blogs/nse_dplyr/nse_dplyr.pdf +0 -0
- data/blogs/oh_my/oh_my.html +0 -568
- data/blogs/ruby_plot/ruby_plot.html +0 -544
- data/blogs/ruby_plot/ruby_plot.pdf +0 -0
- data/examples/latex_templates/Test-acs_article/Test-acs_article.pdf +0 -0
- data/examples/latex_templates/Test-aea_article/Test-aea_article.pdf +0 -0
- data/examples/latex_templates/Test-amq_article/Test-amq_article.pdf +0 -0
- data/examples/latex_templates/Test-amq_article/pics/Figure2.pdf +0 -0
- data/examples/latex_templates/Test-asa_article/Test-asa_article.pdf +0 -0
- data/examples/latex_templates/Test-ieee_article/Test-ieee_article.pdf +0 -0
- data/examples/latex_templates/Test-rjournal_article/RJwrapper.pdf +0 -0
- data/examples/latex_templates/Test-springer_article/Test-springer_article.pdf +0 -0
- data/examples/rmarkdown/svm-rmarkdown-anon-ms-example/svm-rmarkdown-anon-ms-example.pdf +0 -0
- data/examples/rmarkdown/svm-rmarkdown-article-example/svm-rmarkdown-article-example.pdf +0 -0
- data/examples/rmarkdown/svm-rmarkdown-beamer-example/svm-rmarkdown-beamer-example.pdf +0 -0
- data/examples/rmarkdown/svm-rmarkdown-cv/svm-rmarkdown-cv.pdf +0 -0
- data/examples/rmarkdown/svm-rmarkdown-syllabus-example/svm-rmarkdown-syllabus-example.pdf +0 -0
- data/specs/r_dataframe.spec.rb +0 -379
- data/specs/r_environment.spec.rb +0 -140
- data/specs/r_formula.spec.rb +0 -232
- data/specs/r_language.spec.rb +0 -112
- data/specs/r_list.spec.rb +0 -293
- data/specs/r_plots.spec.rb +0 -72
- data/specs/ruby_expression.spec.rb +0 -316
data/blogs/manual/manual.Rmd
CHANGED
|
@@ -1,11 +1,17 @@
|
|
|
1
1
|
---
|
|
2
2
|
title: "Galaaz Manual"
|
|
3
|
-
subtitle: "
|
|
3
|
+
subtitle: "Coupling Ruby (JRuby) and GNU R for data science"
|
|
4
4
|
author: "Rodrigo Botafogo"
|
|
5
|
-
tags: [Galaaz, Ruby, R,
|
|
6
|
-
date: "
|
|
7
|
-
bibliography: "/
|
|
5
|
+
tags: [Galaaz, Ruby, JRuby, R, "GNU R", ggplot2, knitr, dplyr, Bioconductor, Arrow]
|
|
6
|
+
date: "2026"
|
|
7
|
+
bibliography: "../../examples/Bibliography/stats.bib"
|
|
8
8
|
output:
|
|
9
|
+
html_document:
|
|
10
|
+
self_contained: true
|
|
11
|
+
keep_md: true
|
|
12
|
+
toc: true
|
|
13
|
+
toc_depth: 3
|
|
14
|
+
number_sections: true
|
|
9
15
|
pdf_document:
|
|
10
16
|
includes:
|
|
11
17
|
in_header: "../../sty/galaaz.sty"
|
|
@@ -13,15 +19,15 @@ output:
|
|
|
13
19
|
number_sections: yes
|
|
14
20
|
toc: true
|
|
15
21
|
toc_depth: 3
|
|
16
|
-
html_document:
|
|
17
|
-
self_contained: true
|
|
18
|
-
keep_md: true
|
|
19
22
|
md_document:
|
|
20
23
|
variant: markdown_github
|
|
21
24
|
fontsize: 11pt
|
|
22
25
|
---
|
|
23
26
|
|
|
24
27
|
```{ruby setup, echo=FALSE}
|
|
28
|
+
# Bridge default is 60s; some chunks (Arrow, large dplyr pipes) need more.
|
|
29
|
+
ENV['GALAAZ_BRIDGE_TIMEOUT_SEC'] ||= '300'
|
|
30
|
+
|
|
25
31
|
R.options(crayon__enabled: false)
|
|
26
32
|
R.install_and_loads('kableExtra')
|
|
27
33
|
```
|
|
@@ -32,8 +38,9 @@ Galaaz is a system for tightly coupling Ruby and R. Ruby is a powerful language,
|
|
|
32
38
|
community, a very large set of libraries and great for web development. However, it lacks
|
|
33
39
|
libraries for data science, statistics, scientific plotting and machine learning. On the
|
|
34
40
|
other hand, R is considered one of the most powerful languages for solving all of the above
|
|
35
|
-
problems.
|
|
36
|
-
|
|
41
|
+
problems. **Python** is a strong competitor: NumPy, pandas, SciPy, and scikit-learn are
|
|
42
|
+
widely used building blocks, and **PyPI** hosts many thousands of other packages for
|
|
43
|
+
numerical work, machine learning, and beyond.
|
|
37
44
|
|
|
38
45
|
With Galaaz we do not intend to re-implement any of the scientific libraries in R, we allow
|
|
39
46
|
for very tight coupling between the two languages to the point that the Ruby developer does
|
|
@@ -44,59 +51,39 @@ general-purpose programming language. It was designed and developed in the mid-1
|
|
|
44
51
|
"Matz" Matsumoto in Japan." It reached high popularity with the development of Ruby on Rails
|
|
45
52
|
(RoR) by David Heinemeier Hansson. RoR is a web application framework first released
|
|
46
53
|
around 2005. It makes extensive use of Ruby's metaprogramming features. With RoR,
|
|
47
|
-
Ruby became very popular. According to [Ruby
|
|
48
|
-
it
|
|
49
|
-
|
|
50
|
-
most popular language.
|
|
54
|
+
Ruby became very popular. According to [Ruby’s place in the TIOBE index](https://www.tiobe.com/tiobe-index/ruby/)
|
|
55
|
+
it peaked in popularity around 2008, then declined until 2015 when it started picking up again.
|
|
56
|
+
Ruby remains a significant language in web development and general-purpose scripting.
|
|
51
57
|
|
|
52
58
|
Python, a language similar to Ruby, ranks 4th in the index. Java, C and C++ take the
|
|
53
59
|
first three positions. Ruby is often criticized for its focus on web applications.
|
|
54
60
|
But Ruby can do [much more](https://github.com/markets/awesome-ruby) than just web applications.
|
|
55
|
-
Yet, for scientific computing, Ruby lags
|
|
56
|
-
|
|
61
|
+
Yet, for scientific computing, Ruby lags behind Python and R. Python offers Django and
|
|
62
|
+
similar frameworks for the web, plus NumPy, pandas, and a deep catalog of science and ML libraries.
|
|
57
63
|
R is a free software environment for statistical computing and graphics with thousands
|
|
58
64
|
of libraries for data analysis.
|
|
59
65
|
|
|
60
66
|
Until recently, there was no real perspective for Ruby to bridge this gap.
|
|
61
67
|
Implementing a complete scientific computing infrastructure would take too long.
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
> * That library is not available in my language. I need to rewrite it.
|
|
82
|
-
> * That language would be the perfect fit for my problem, but we cannot
|
|
83
|
-
> run it in our environment.
|
|
84
|
-
> * That problem is already solved in my language, but the language is
|
|
85
|
-
> too slow.
|
|
86
|
-
>
|
|
87
|
-
> With GraalVM we aim to allow developers to freely choose the right language for
|
|
88
|
-
> the task at hand without making compromises.
|
|
89
|
-
|
|
90
|
-
As stated above, GraalVM is a _universal_ virtual machine that allows Ruby and R (and other
|
|
91
|
-
languages) to run on the same environment. GraalVM allows polyglot applications to
|
|
92
|
-
_seamlessly_ interact with one another and pass values from one language to the other.
|
|
93
|
-
Although a great idea, GraalVM still requires application writers to know several languages.
|
|
94
|
-
To eliminate that requirement, we built Galaaz, a gem for Ruby, to tightly couple
|
|
95
|
-
Ruby and R and allow those languages to interact in a way that the user will be unaware
|
|
96
|
-
of such interaction. In other words, a Ruby programmer will be able to use all
|
|
97
|
-
the capabilities of R without knowing the R syntax.
|
|
98
|
-
|
|
99
|
-
Library wrapping is a usual way of bringing features from one language into another.
|
|
68
|
+
|
|
69
|
+
**Galaaz 2.0** couples **JRuby** (Ruby on the JVM) with **GNU R**—the same R you use for
|
|
70
|
+
CRAN and Bioconductor. Ruby and R run in **separate processes**; the **Galaaz bridge**
|
|
71
|
+
sends requests to R and returns results to Ruby. From your point of view you still write
|
|
72
|
+
Ruby: `R.c(...)`, `R.library('ggplot2')`, `~R[:mtcars]`, and dplyr-style chains on R objects.
|
|
73
|
+
You do not need to learn R syntax to get a lot done, though reading R documentation for
|
|
74
|
+
individual packages remains useful.
|
|
75
|
+
|
|
76
|
+
Earlier experiments with Galaaz used Oracle’s **GraalVM** with TruffleRuby and FastR so that
|
|
77
|
+
Ruby and R could share one runtime. That path is no longer the focus: **standard GNU R**
|
|
78
|
+
gives full compatibility with the R package ecosystem (including compiled extensions and
|
|
79
|
+
Bioconductor) while JRuby gives a mature Ruby with **real multithreading** for application
|
|
80
|
+
and I/O code.
|
|
81
|
+
|
|
82
|
+
The bridge handles **communication and typing** between the two worlds; large tables can
|
|
83
|
+
also flow through **Apache Arrow** on the R side when you use the optional helpers described
|
|
84
|
+
later in this manual.
|
|
85
|
+
|
|
86
|
+
Library wrapping is a common way to bring features from one language into another.
|
|
100
87
|
To improve performance, Python often wraps more efficient C libraries. For the
|
|
101
88
|
Python developer, the existence of such C libraries is hidden. The problem with
|
|
102
89
|
library wrapping is that for any new library, there is the need to handcraft a new
|
|
@@ -121,27 +108,202 @@ Galaaz is the Portuguese name for "Galahad". From Wikipedia:
|
|
|
121
108
|
His name should not be mistaken with Galehaut, a different knight from
|
|
122
109
|
Arthurian legend.
|
|
123
110
|
|
|
111
|
+
# Command-line tools (`bin/`)
|
|
112
|
+
|
|
113
|
+
The Galaaz repository ships many helpers under **`bin/`**. When working from a **clone**, call
|
|
114
|
+
them as **`bin/<name>`** from the project root (or `./bin/<name>`). If you install the **gem**,
|
|
115
|
+
only a subset is guaranteed on your `PATH` (see the gemspec: **`galaaz`**, **`gstudio`**, **`gknit`**, **`grun`**, **`gknit-draft`**); for development and CI, prefer the **`bin/`** copies so JVM flags and paths stay correct.
|
|
116
|
+
|
|
117
|
+
Below, **current (Galaaz 2.0 + JRuby + GNU R)** means the tool is wired to **`jruby`** and
|
|
118
|
+
**`bin/galaaz_jruby_env.inc.sh`** (or equivalent logic in Ruby via `lib/galaaz_jruby.rb`). **Legacy**
|
|
119
|
+
means the script still targets **GraalVM** polyglot Ruby / FastR-era invocation and is **not**
|
|
120
|
+
expected to work on a typical JRuby-only setup.
|
|
121
|
+
|
|
122
|
+
**Table layout:** names in the first column are **`bin/`** filenames (run as `bin/<name>` from the repo root). Long options and examples sit **outside** the tables so PDF columns stay readable.
|
|
123
|
+
|
|
124
|
+
```{r bin-tables-helper, echo=FALSE}
|
|
125
|
+
bin_tbl <- function(df) {
|
|
126
|
+
k <- knitr::kable(df, row.names = FALSE, booktabs = TRUE, linesep = "",
|
|
127
|
+
col.names = c("Script", "Role", "2.0?"))
|
|
128
|
+
if (knitr::is_latex_output()) {
|
|
129
|
+
k <- kableExtra::kable_styling(k, font_size = 9, latex_options = "scale_down")
|
|
130
|
+
k <- kableExtra::column_spec(k, 1, width = "2.5cm")
|
|
131
|
+
k <- kableExtra::column_spec(k, 2, width = "9.5cm")
|
|
132
|
+
k <- kableExtra::column_spec(k, 3, width = "2.8cm")
|
|
133
|
+
} else {
|
|
134
|
+
k <- kableExtra::kable_styling(k, bootstrap_options = c("striped", "condensed"), full_width = TRUE)
|
|
135
|
+
}
|
|
136
|
+
k
|
|
137
|
+
}
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
```{r bin-tables-bootstrap, echo=FALSE}
|
|
141
|
+
df_boot <- data.frame(
|
|
142
|
+
Script = c("galaaz-bootstrap", "galaaz-jruby", "galaaz_jruby_env.inc.sh", "install-tinytex"),
|
|
143
|
+
Role = c(
|
|
144
|
+
"WSL2 helper: Docker checks; optional TinyTeX or poppler for gKnit PDF.",
|
|
145
|
+
"JRuby with repo lib/ on LOAD_PATH and required JVM flags (e.g. Arrow).",
|
|
146
|
+
"Sourced by bash wrappers; sets GALAAZ_REQUIRED_JRUBY_J_ARGS.",
|
|
147
|
+
"Install TinyTeX for PDF output."
|
|
148
|
+
),
|
|
149
|
+
X2 = c("Yes*", "Yes", "Yes†", "Yes"),
|
|
150
|
+
stringsAsFactors = FALSE
|
|
151
|
+
)
|
|
152
|
+
bin_tbl(df_boot)
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
\* Where WSL/Docker apply. **`galaaz-bootstrap` flags:** `--check`, `--apply`, `--runtime` (`docker` \| `local` \| `auto`), `--[no-]prompt-doc-tools`.
|
|
156
|
+
|
|
157
|
+
† Not run directly.
|
|
158
|
+
|
|
159
|
+
**`galaaz-jruby` examples** (from repo root):
|
|
160
|
+
|
|
161
|
+
```text
|
|
162
|
+
bin/galaaz-jruby my_script.rb
|
|
163
|
+
bin/galaaz-jruby -S rspec
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## Interactive use, examples, and Rake
|
|
167
|
+
|
|
168
|
+
```{r bin-tables-interactive, echo=FALSE}
|
|
169
|
+
df_ix <- data.frame(
|
|
170
|
+
Script = c("gstudio", "run_example", "galaaz"),
|
|
171
|
+
Role = c(
|
|
172
|
+
"IRB or Pry with Galaaz preloaded (JRuby + JVM flags).",
|
|
173
|
+
"Run one Ruby file using the same JRuby/JVM setup as tests.",
|
|
174
|
+
"Forward arguments to rake (needs rake; usually JRuby)."
|
|
175
|
+
),
|
|
176
|
+
X2 = c("Yes", "Yes", "Yes"),
|
|
177
|
+
stringsAsFactors = FALSE
|
|
178
|
+
)
|
|
179
|
+
bin_tbl(df_ix)
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
## gKnit and document drafts
|
|
183
|
+
|
|
184
|
+
```{r bin-tables-gknit, echo=FALSE}
|
|
185
|
+
df_gk <- data.frame(
|
|
186
|
+
Script = c("gknit", "gknit-draft", "gknit-draft.rb", "gknit_Rscript"),
|
|
187
|
+
Role = c(
|
|
188
|
+
"Knit .Rmd via JRuby and R Markdown render.",
|
|
189
|
+
"Drafts from rticles-style templates; wrapper still uses legacy polyglot ruby.",
|
|
190
|
+
"Ruby entry: GKnit.draft (use with JRuby + LOAD_PATH).",
|
|
191
|
+
"Polyglot Rscript launcher; hard-coded LOAD_PATH sample."
|
|
192
|
+
),
|
|
193
|
+
X2 = c("Yes", "Legacy", "JRuby", "No"),
|
|
194
|
+
stringsAsFactors = FALSE
|
|
195
|
+
)
|
|
196
|
+
bin_tbl(df_gk)
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
**`gknit` CLI** (see `gknit -h`): `--output_format`, `--output_file`, `--output_dir`, `--bridge_timeout_sec`, `--callback_timeout_ms`. If `--output_format` is omitted, the **first** YAML `output:` target wins.
|
|
200
|
+
|
|
201
|
+
Prefer **`galaaz-jruby`** for **`gknit-draft`** workflows until that wrapper matches the **`gknit`** stack.
|
|
202
|
+
|
|
203
|
+
## Tests
|
|
204
|
+
|
|
205
|
+
```{r bin-tables-tests, echo=FALSE}
|
|
206
|
+
df_ts <- data.frame(
|
|
207
|
+
Script = c("run_rspec", "run_all_rspec", "run_slow_rspec", "run_old_rspec", "run_rspec_subset"),
|
|
208
|
+
Role = c(
|
|
209
|
+
"Top-level specs/*_spec.rb with spec_helper (see docs/testing.md).",
|
|
210
|
+
"Compile ext/new_bridge; run specs/ and new_bridge_specs/ together.",
|
|
211
|
+
"Suites under slow-specs/ (read script header for spec_helper).",
|
|
212
|
+
"Legacy suites under old_specs/.",
|
|
213
|
+
"Numbered subset 1–18 (Documentation/Spec_Subsets.md)."
|
|
214
|
+
),
|
|
215
|
+
X2 = c("Yes", "Yes", "Yes", "Yes", "Yes"),
|
|
216
|
+
stringsAsFactors = FALSE
|
|
217
|
+
)
|
|
218
|
+
bin_tbl(df_ts)
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
## Other
|
|
222
|
+
|
|
223
|
+
```{r bin-tables-other, echo=FALSE}
|
|
224
|
+
df_ot <- data.frame(
|
|
225
|
+
Script = c("grun", "gstudio_irb.rb / gstudio_pry.rb"),
|
|
226
|
+
Role = c(
|
|
227
|
+
"Graal-era launcher: polyglot ruby with --jvm. Use galaaz-jruby -S instead.",
|
|
228
|
+
"Loaded by gstudio; not meant to be run standalone."
|
|
229
|
+
),
|
|
230
|
+
X2 = c("No", "Yes"),
|
|
231
|
+
stringsAsFactors = FALSE
|
|
232
|
+
)
|
|
233
|
+
bin_tbl(df_ot)
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
For day-to-day **2.0** use, rely on **`bin/galaaz-jruby`**, **`bin/gstudio`**, **`bin/gknit`**, **`bin/run_example`**, **`bin/run_rspec`** / **`bin/run_all_rspec`**, and **`bin/galaaz-bootstrap`** on WSL when using Dockerized R. Treat **`grun`**, **`gknit_Rscript`**, and the polyglot **`ruby`** invocation in **`gknit-draft`** as **legacy** until they are ported to the same JRuby path as **`gknit`**.
|
|
237
|
+
|
|
124
238
|
# System Compatibility
|
|
125
239
|
|
|
126
|
-
|
|
127
|
-
* Ubuntu 18.04 LTS
|
|
128
|
-
* Ubuntu 16.04 LTS
|
|
129
|
-
* Fedora 28
|
|
130
|
-
* macOS 10.14 (Mojave)
|
|
131
|
-
* macOS 10.13 (High Sierra)
|
|
240
|
+
Typical development and CI targets:
|
|
132
241
|
|
|
133
|
-
|
|
242
|
+
* **Linux** — recent Ubuntu LTS or comparable distributions (x86_64).
|
|
243
|
+
* **macOS** — recent releases with JRuby and GNU R available.
|
|
244
|
+
* **Windows** — use **WSL2** (same Linux stack as above); native Windows is not the primary target.
|
|
134
245
|
|
|
135
|
-
|
|
136
|
-
* FastR
|
|
246
|
+
The native **gatekeeper** component under `ext/new_bridge` is built with `make` and a C++ toolchain; see the project `README` if compilation fails on your platform.
|
|
137
247
|
|
|
248
|
+
# Dependencies
|
|
249
|
+
|
|
250
|
+
* **JRuby** — Galaaz 2.0 requires JRuby (tested with **10.1.1.0**) and a matching **JDK** (tested with **Java 21**). MRI Ruby is not supported.
|
|
251
|
+
* **GNU R** — `R` and `Rscript` on your `PATH` (tested with **4.3.3**), plus a C++ toolchain (`g++`, `make`) and the **Rcpp** package to compile the gatekeeper.
|
|
252
|
+
* **galaaz gem** — runtime dependency `msgpack` is pulled in by `gem install`.
|
|
253
|
+
* Optional: **Docker** — if you run R in a container (common on WSL2); see bootstrap below.
|
|
254
|
+
* Optional R packages for examples in this manual — e.g. `ggplot2`, `dplyr`, `knitr`, `kableExtra`, `arrow`, Bioconductor tools such as **DESeq2** (installed the usual R way).
|
|
138
255
|
|
|
139
256
|
# Installation
|
|
140
257
|
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
258
|
+
The supported install is **`gem install` + compile the gatekeeper**. You do not need a git clone.
|
|
259
|
+
|
|
260
|
+
1. Install **JRuby**, a compatible **JDK**, and **GNU R** (with `Rscript` and a C++ compiler).
|
|
261
|
+
2. In R, install **Rcpp**: `install.packages("Rcpp")`.
|
|
262
|
+
3. Install the gem: `jruby -S gem install galaaz`
|
|
263
|
+
4. Compile the native gatekeeper from the installed gem:
|
|
264
|
+
|
|
265
|
+
```
|
|
266
|
+
gem_dir="$(jruby -e "puts Gem::Specification.find_by_name('galaaz').full_gem_path")"
|
|
267
|
+
make -C "${gem_dir}/ext/new_bridge" all
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
5. Ensure **`R`** starts GNU R and can install packages (network access to CRAN when you first call `R.install_and_loads`). For **Apache Arrow** on Java 9+, pass `-J--add-opens=java.base/java.nio=ALL-UNNAMED` to JRuby (from a checkout, `bin/galaaz-jruby` does this).
|
|
271
|
+
|
|
272
|
+
For **gKnit**, **knitr**, **rmarkdown**, and LaTeX (PDF output), install the corresponding R packages, **Pandoc**, and a TeX distribution if you need PDF; the repository includes helpers such as **`bin/install-tinytex`** where appropriate.
|
|
273
|
+
|
|
274
|
+
A **table of all `bin/` scripts** (bootstrap, JRuby wrapper, gstudio, gknit, test runners, and which ones are legacy) is in the section **Command-line tools (`bin/`)** earlier in this manual.
|
|
275
|
+
|
|
276
|
+
### From a repository checkout (contributors)
|
|
277
|
+
|
|
278
|
+
1. Install **bundler** if needed, then run **`jruby -S bundle install`** in the repository root.
|
|
279
|
+
2. Build the bridge native code: **`make -C ext/new_bridge all`** (or **`rake compile_gatekeeper`**).
|
|
280
|
+
3. Run scripts with **`bin/galaaz-jruby`** (sources **`bin/galaaz_jruby_env.inc.sh`** and adds **`-I lib`**).
|
|
281
|
+
|
|
282
|
+
Maintainers can prove a built `.gem` on a throwaway Ubuntu machine (no repo inside the container) with **`./docker/cold-install/run.sh`**.
|
|
283
|
+
|
|
284
|
+
## Windows + WSL2 (optional: Docker / R in a container)
|
|
285
|
+
|
|
286
|
+
If you run Galaaz on Windows through WSL2 and want containerized R instances,
|
|
287
|
+
Docker Desktop is the supported setup.
|
|
288
|
+
|
|
289
|
+
1. Install Docker Desktop on Windows:
|
|
290
|
+
- https://www.docker.com/products/docker-desktop/
|
|
291
|
+
2. Open Docker Desktop and enable WSL integration:
|
|
292
|
+
- Settings > Resources > WSL Integration
|
|
293
|
+
- Enable integration for your target distro
|
|
294
|
+
- Apply & Restart Docker Desktop
|
|
295
|
+
3. In WSL, run Galaaz bootstrap:
|
|
296
|
+
|
|
297
|
+
> ruby bin/galaaz-bootstrap --apply
|
|
298
|
+
> ruby bin/galaaz-bootstrap --check
|
|
299
|
+
|
|
300
|
+
Expected result:
|
|
301
|
+
- docker CLI available
|
|
302
|
+
- docker compose available
|
|
303
|
+
- docker daemon reachable (`docker info` works)
|
|
304
|
+
|
|
305
|
+
If bootstrap reports daemon is unreachable, check Docker Desktop is running and
|
|
306
|
+
WSL integration is enabled for the distro where Galaaz is installed.
|
|
145
307
|
|
|
146
308
|
# Usage
|
|
147
309
|
|
|
@@ -170,7 +332,7 @@ Galaaz is the Portuguese name for "Galahad". From Wikipedia:
|
|
|
170
332
|
|
|
171
333
|
> galaaz -T
|
|
172
334
|
|
|
173
|
-
Shows a list with all available
|
|
335
|
+
Shows a list with all available executable tasks. To execute a task, substitute the
|
|
174
336
|
'rake' word in the list with 'galaaz'. For instance, the following line shows up
|
|
175
337
|
after 'galaaz -T'
|
|
176
338
|
|
|
@@ -180,20 +342,228 @@ Galaaz is the Portuguese name for "Galahad". From Wikipedia:
|
|
|
180
342
|
|
|
181
343
|
> galaaz master_list:scatter_plot
|
|
182
344
|
|
|
345
|
+
# JRuby, multithreading, and the R bridge
|
|
346
|
+
|
|
347
|
+
Galaaz 2.0 runs Ruby on **JRuby**, so your application can use **real parallel threads** for
|
|
348
|
+
I/O-bound work (HTTP clients, database connections, message consumers, and so on). R itself is
|
|
349
|
+
still executed in a **single GNU R process** behind the Galaaz bridge.
|
|
350
|
+
|
|
351
|
+
When several Ruby threads call into R at the same time, the bridge **serializes** those calls:
|
|
352
|
+
each request is matched to a reply using an internal per-call **queue**, so you do not need to
|
|
353
|
+
add your own mutex around every `R.foo` from application threads. (You should still use normal
|
|
354
|
+
Ruby synchronization when **Ruby** data structures are shared between threads—for example, when
|
|
355
|
+
appending rows from each thread into a shared array before sending them to R.)
|
|
356
|
+
|
|
357
|
+
A practical pattern is:
|
|
358
|
+
|
|
359
|
+
1. Use threads (or a connection pool) to read from **multiple databases or shards** in parallel.
|
|
360
|
+
2. Merge the rows in Ruby under a `Mutex` if you collect into one structure.
|
|
361
|
+
3. Hand the merged table to R **once** (for example with `R::Arrow.from_ruby_batches` and dplyr,
|
|
362
|
+
or by building a data frame) so heavy statistics run in R with fewer bridge round-trips.
|
|
363
|
+
|
|
364
|
+
A runnable sketch lives in
|
|
365
|
+
`examples/multithread_shards_to_r/shards_to_r.rb` (simulated shard queries; swap in your DB
|
|
366
|
+
driver). For concurrency tests on the bridge itself, see `specs/bridge_concurrent_spec.rb` and
|
|
367
|
+
`specs/arrow_from_ruby_batches_spec.rb`.
|
|
368
|
+
|
|
369
|
+
## Long-running R calls and a completion block
|
|
370
|
+
|
|
371
|
+
For R work that can take a long time, the bridge can avoid a Ruby-side **wait timeout** by
|
|
372
|
+
scheduling the call and resuming in a **block** when the `RET` arrives.
|
|
373
|
+
|
|
374
|
+
- **`R.eval_r_async(code, timeout: nil) { |result| ... }`** — string eval; on success, `result.value`
|
|
375
|
+
is the same formatted string as **`R.eval_r`** (use `timeout: nil` for no Ruby-side limit).
|
|
376
|
+
- **`R::Async.<rname>(...) { |result| ... }`** — same dispatch as **`R.<rname>(...)`**, but async;
|
|
377
|
+
on success, `result.value` is an **`R::Object`** (or unboxed Ruby value / Symbol), like synchronous
|
|
378
|
+
**`R.<rname>`**. Optional keyword **`timeout:`** applies a Ruby-side wait limit (completion receives
|
|
379
|
+
**`NewBridge::SessionClient::TimeoutError`** if R is too slow).
|
|
380
|
+
|
|
381
|
+
**Important:** **`R.foo(...) { |x| }`** is already used for dplyr-style scopes (`R::Support.new_scope`),
|
|
382
|
+
so async R calls must use **`R::Async`** or **`R.eval_r_async`**, not a bare **`R.foo` with a block.**
|
|
383
|
+
|
|
384
|
+
`NewBridge::EvalResult` exposes **`#ok?`**, **`#value`**, and **`#error`**. The completion block runs on a
|
|
385
|
+
**background thread** (not the bridge reader thread).
|
|
386
|
+
|
|
387
|
+
The example below is **plain Ruby** (no Rails). The R snippet sleeps (standing in for heavy work) and then
|
|
388
|
+
returns an integer so the success branch shows a **non-nil** value. (`Sys.sleep` alone returns **NULL** in R;
|
|
389
|
+
on success **`result.value`** is then **`nil`** in Ruby—that is expected, not a bridge error.)
|
|
390
|
+
|
|
391
|
+
```{ruby long_r_completion_block}
|
|
392
|
+
require 'thread'
|
|
393
|
+
|
|
394
|
+
completion = Queue.new
|
|
395
|
+
|
|
396
|
+
R.eval_r_async('({ Sys.sleep(0.3); 42L })', timeout: nil) do |result|
|
|
397
|
+
if result.ok?
|
|
398
|
+
puts "[completion] R finished; eval_r-style value: #{result.value.inspect}"
|
|
399
|
+
else
|
|
400
|
+
puts "[completion] R/bridge error: #{result.error.class}: #{result.error.message}"
|
|
401
|
+
end
|
|
402
|
+
completion.push(:done)
|
|
403
|
+
end
|
|
404
|
+
|
|
405
|
+
3.times do |i|
|
|
406
|
+
puts "[main] other Ruby work step #{i + 1}"
|
|
407
|
+
sleep 0.05
|
|
408
|
+
end
|
|
409
|
+
|
|
410
|
+
completion.pop
|
|
411
|
+
puts "[main] R completion has run; exiting."
|
|
412
|
+
```
|
|
413
|
+
|
|
414
|
+
In a **web application**, the HTTP response usually ends before R finishes, so you would not
|
|
415
|
+
`Queue#pop` in the controller; you would persist an identifier, let the completion block write
|
|
416
|
+
the outcome to storage, and notify the client (poll, WebSocket, Turbo Stream, etc.). The plain
|
|
417
|
+
Ruby pattern above is only to show **when** the result exists (inside the block, or after data
|
|
418
|
+
written there is observed elsewhere). Runnable specs live in **`new_bridge_specs/eval_r_async_spec.rb`**.
|
|
419
|
+
|
|
420
|
+
## Galaaz + Rails (JRuby) integration baseline
|
|
421
|
+
|
|
422
|
+
This section documents the baseline we used to create a working Rails app with Galaaz in WSL.
|
|
423
|
+
The goals were:
|
|
424
|
+
|
|
425
|
+
1. Rails boots under **JRuby**.
|
|
426
|
+
2. Galaaz is loaded from a local checkout (before publishing to RubyGems).
|
|
427
|
+
3. A request path can execute **`R.eval(...)`** and return a result.
|
|
428
|
+
|
|
429
|
+
### 1) Create the app with JRuby-friendly options
|
|
430
|
+
|
|
431
|
+
Rails defaults can pull gems that are not ideal on JRuby-first setups (for example sqlite native
|
|
432
|
+
extension paths and deployment extras). A minimal app avoids early friction:
|
|
433
|
+
|
|
434
|
+
```bash
|
|
435
|
+
cd /home/rbotafogo/desenv_linux
|
|
436
|
+
jruby -S rails new hedi --skip-git --minimal --skip-kamal --skip-solid --skip-active-record
|
|
437
|
+
```
|
|
438
|
+
|
|
439
|
+
Then install gems:
|
|
440
|
+
|
|
441
|
+
```bash
|
|
442
|
+
cd /home/rbotafogo/desenv_linux/hedi
|
|
443
|
+
jruby -S bundle install
|
|
444
|
+
```
|
|
445
|
+
|
|
446
|
+
### 2) Use Galaaz as a local path gem
|
|
447
|
+
|
|
448
|
+
For local development we keep a stable path:
|
|
449
|
+
|
|
450
|
+
- `~/gems/galaaz` -> symlink to your Galaaz checkout
|
|
451
|
+
- optional built gem archive in `~/gems/pkg/`
|
|
452
|
+
|
|
453
|
+
In Rails `Gemfile`:
|
|
454
|
+
|
|
455
|
+
```ruby
|
|
456
|
+
gem "galaaz", path: "/home/rbotafogo/gems/galaaz", require: false
|
|
457
|
+
```
|
|
458
|
+
|
|
459
|
+
And load after Rails boot in `config/application.rb`:
|
|
460
|
+
|
|
461
|
+
```ruby
|
|
462
|
+
config.after_initialize { require "galaaz" }
|
|
463
|
+
```
|
|
464
|
+
|
|
465
|
+
Why `require: false` + `after_initialize`? In this integration, loading Galaaz too early via
|
|
466
|
+
`Bundler.require` triggered Rails/JRuby initialization failures.
|
|
467
|
+
|
|
468
|
+
### 3) Simple request-path smoke test
|
|
469
|
+
|
|
470
|
+
A direct smoke test from Rails runner:
|
|
471
|
+
|
|
472
|
+
```bash
|
|
473
|
+
cd /home/rbotafogo/desenv_linux/hedi
|
|
474
|
+
jruby -S bundle exec rails runner "puts R.eval('sum(c(1,2,3,4,5))').inspect"
|
|
475
|
+
```
|
|
476
|
+
|
|
477
|
+
Expected output:
|
|
478
|
+
|
|
479
|
+
```text
|
|
480
|
+
15.0
|
|
481
|
+
```
|
|
482
|
+
|
|
483
|
+
### 4) HTTP endpoint pattern
|
|
484
|
+
|
|
485
|
+
For this baseline, a small Rack endpoint was the most stable first step to prove request-time R
|
|
486
|
+
evaluation. (A full ActionController stack can be enabled later as the app evolves.)
|
|
487
|
+
|
|
488
|
+
Minimal pattern:
|
|
489
|
+
|
|
490
|
+
1. Define a Rack app class under `lib/` that runs `R.eval(...)` and returns HTML/JSON.
|
|
491
|
+
2. Point a route to that Rack app (`root to: MyRackApp`).
|
|
492
|
+
3. Verify with browser/curl.
|
|
493
|
+
|
|
494
|
+
### 5) Running from WSL and opening from Windows
|
|
495
|
+
|
|
496
|
+
Recommended bind:
|
|
497
|
+
|
|
498
|
+
```bash
|
|
499
|
+
jruby -S bundle exec rails server -b 0.0.0.0 -p 3000
|
|
500
|
+
```
|
|
501
|
+
|
|
502
|
+
Then open from Windows:
|
|
503
|
+
|
|
504
|
+
- `http://localhost:3000` (usually works with WSL localhost forwarding), or
|
|
505
|
+
- `http://<wsl-ip>:3000` if needed.
|
|
506
|
+
|
|
507
|
+
In development, if Host Authorization blocks requests with unexpected Host headers, use:
|
|
508
|
+
|
|
509
|
+
```ruby
|
|
510
|
+
# config/environments/development.rb
|
|
511
|
+
config.hosts.clear
|
|
512
|
+
```
|
|
513
|
+
|
|
514
|
+
### 6) Troubleshooting checklist
|
|
515
|
+
|
|
516
|
+
- `Could not find ... in locally installed gems`:
|
|
517
|
+
run `jruby -S bundle install` in the Rails app directory.
|
|
518
|
+
- Stale PID after crash:
|
|
519
|
+
remove `tmp/pids/server.pid`.
|
|
520
|
+
- Local Galaaz path changed:
|
|
521
|
+
verify `Gemfile` path target exists and rerun bundler.
|
|
522
|
+
- R runtime issues:
|
|
523
|
+
confirm GNU R is installed and on `PATH` in the same shell where Rails runs.
|
|
524
|
+
|
|
525
|
+
As new Rails features are added (controllers, jobs, websockets, background rendering, plot
|
|
526
|
+
generation), extend this section with concrete, runnable snippets and the associated operational
|
|
527
|
+
checks.
|
|
183
528
|
|
|
184
529
|
# Accessing R from Ruby
|
|
185
530
|
|
|
186
|
-
One of the nice aspects of Galaaz
|
|
187
|
-
be easily accessed from Ruby. For instance, to access the
|
|
188
|
-
in Ruby, we use the
|
|
189
|
-
value of the
|
|
531
|
+
One of the nice aspects of Galaaz is that variables and functions defined in R can
|
|
532
|
+
be easily accessed from Ruby. For instance, to access the `mtcars` data frame from R
|
|
533
|
+
in Ruby, we use the symbol `:mtcars` preceded by the `~` operator: `~R[:mtcars]` retrieves the
|
|
534
|
+
value of the `mtcars` object in R.
|
|
190
535
|
|
|
191
536
|
```{ruby access_r}
|
|
192
|
-
puts
|
|
537
|
+
puts ~R[:mtcars]
|
|
538
|
+
```
|
|
539
|
+
|
|
540
|
+
## Scoped symbols and lexical scoping
|
|
541
|
+
|
|
542
|
+
Galaaz 2.0 uses **scoped symbols** by default. The canonical style is `R[:name]`:
|
|
543
|
+
|
|
544
|
+
- `~R[:mtcars]` fetches an R object by name.
|
|
545
|
+
- `R[:a] + R[:b]` builds an expression.
|
|
546
|
+
- `R[:year].up_to(R[:day])` builds range expressions.
|
|
547
|
+
|
|
548
|
+
If you prefer the terse `:x` syntax, you can opt in with lexical scoping using a Ruby refinement:
|
|
549
|
+
|
|
550
|
+
```ruby
|
|
551
|
+
module MyScript
|
|
552
|
+
using Galaaz::SymbolDSL
|
|
553
|
+
|
|
554
|
+
def self.run
|
|
555
|
+
expr = :a + :b
|
|
556
|
+
puts expr
|
|
557
|
+
puts ~:mtcars
|
|
558
|
+
end
|
|
559
|
+
end
|
|
193
560
|
```
|
|
194
561
|
|
|
195
|
-
|
|
196
|
-
|
|
562
|
+
`using Galaaz::SymbolDSL` is **lexically scoped**: only code in that module/file scope gets `:x` DSL behavior.
|
|
563
|
+
Outside that scope, plain Ruby `Symbol` behavior is unchanged.
|
|
564
|
+
|
|
565
|
+
To access an R function from Ruby, the R function needs to be preceded by `R.` scoping.
|
|
566
|
+
Below we see an example of creating a R::Vector by calling the 'c' R function
|
|
197
567
|
|
|
198
568
|
```{ruby call_r_func}
|
|
199
569
|
puts vec = R.c(1.0, 2.0, 3.0, 4.0)
|
|
@@ -224,7 +594,7 @@ The call above to the 'c' function can also be done using '.' notation:
|
|
|
224
594
|
```{ruby concat_with_dot}
|
|
225
595
|
puts vec.c(10, 20, 30)
|
|
226
596
|
```
|
|
227
|
-
We will talk about vector indexing in a
|
|
597
|
+
We will talk about vector indexing in a later section. But notice here that indexing
|
|
228
598
|
an R::Vector will return another R::Vector:
|
|
229
599
|
|
|
230
600
|
```{ruby indexing}
|
|
@@ -258,12 +628,12 @@ puts vec.map { |x| x + 2 }
|
|
|
258
628
|
|
|
259
629
|
# gKnitting a Document
|
|
260
630
|
|
|
261
|
-
This manual has been formatted
|
|
262
|
-
a document in Ruby or R and output it in any of the available formats for R
|
|
263
|
-
gKnit runs
|
|
631
|
+
This manual has been formatted using gKnit. gKnit uses knitr and R Markdown to knit
|
|
632
|
+
a document in Ruby or R and output it in any of the available formats for R Markdown.
|
|
633
|
+
gKnit runs with **JRuby**, **GNU R**, and Galaaz. In gKnit, Ruby variables are persisted between
|
|
264
634
|
chunks, making it an ideal solution for literate programming. Also, since it is based
|
|
265
|
-
on Galaaz, Ruby chunks can have access to R variables and
|
|
266
|
-
|
|
635
|
+
on Galaaz, Ruby chunks can have access to R variables and combining Ruby with R in one
|
|
636
|
+
document is natural.
|
|
267
637
|
|
|
268
638
|
The idea of "literate programming" was first introduced by Donald Knuth in the
|
|
269
639
|
1980's [@Knuth:literate_programming].
|
|
@@ -282,7 +652,7 @@ single document or set of documents that when distributed to peers could be reru
|
|
|
282
652
|
the same output and reports.
|
|
283
653
|
|
|
284
654
|
The R community has put a great deal of effort in reproducible research. In 2002, Sweave was
|
|
285
|
-
introduced and it allowed mixing R code with
|
|
655
|
+
introduced and it allowed mixing R code with LaTeX, generating high-quality PDF documents. A
|
|
286
656
|
Sweave document could include code, the results of executing the code, graphics and text
|
|
287
657
|
such that it contained the whole narrative to reproduce the research. In
|
|
288
658
|
2012, Knitr, developed by Yihui Xie from RStudio was released to replace Sweave and to
|
|
@@ -291,7 +661,7 @@ were necessary for Sweave.
|
|
|
291
661
|
|
|
292
662
|
With Knitr, __R markdown__ was also developed, an extension to the
|
|
293
663
|
Markdown format. With __R markdown__ and Knitr it is possible to generate reports in a multitude
|
|
294
|
-
of formats such as HTML,
|
|
664
|
+
of formats such as HTML, Markdown, LaTeX, PDF, DVI, etc. __R markdown__ also allows the use of
|
|
295
665
|
multiple programming languages such as R, Ruby, Python, etc. in the same document.
|
|
296
666
|
|
|
297
667
|
In __R markdown__, text is interspersed with
|
|
@@ -326,7 +696,7 @@ Now, any single code has dozens of variables that we might want to use and reuse
|
|
|
326
696
|
Clearly, such an approach becomes quickly unmanageable. Probably, because of
|
|
327
697
|
this problem, it is very rare to see any __R markdown__ document in the Ruby community.
|
|
328
698
|
|
|
329
|
-
When variables can be used
|
|
699
|
+
When variables can be used across chunks, then no overhead is needed:
|
|
330
700
|
|
|
331
701
|
```{ruby persistence}
|
|
332
702
|
lst = R.list(a: 1, b: 2, c: 3)
|
|
@@ -338,8 +708,8 @@ puts lst
|
|
|
338
708
|
```
|
|
339
709
|
|
|
340
710
|
In the Python community, the same effort to have code and text in an integrated environment
|
|
341
|
-
started around the first decade of
|
|
342
|
-
Fernando Pérez
|
|
711
|
+
started around the first decade of the 2000s. In 2006 IPython 0.7.2 was released. In 2014,
|
|
712
|
+
Fernando Pérez spun off the Jupyter project from IPython, creating a web-based interactive
|
|
343
713
|
computation environment. Jupyter can now be used with many languages, including Ruby with the
|
|
344
714
|
iruby gem (https://github.com/SciRuby/iruby). In order to have multiple languages in a Jupyter
|
|
345
715
|
notebook the SoS kernel was developed (https://vatlab.github.io/sos-docs/).
|
|
@@ -353,8 +723,8 @@ have in a single document, text and code.
|
|
|
353
723
|
|
|
354
724
|
In gKnit, Ruby variables are persisted between
|
|
355
725
|
chunks, making it an ideal solution for literate programming in this language. Also,
|
|
356
|
-
since it is based on
|
|
357
|
-
|
|
726
|
+
since it is based on Galaaz, Ruby chunks can access R variables (`~R[:name]`, `R.*`) through the
|
|
727
|
+
**Galaaz bridge** while knitr drives **GNU R**—no GraalVM polyglot runtime is required.
|
|
358
728
|
|
|
359
729
|
This is not a blog post on __R markdown__, and the interested user is directed to the following links
|
|
360
730
|
for detailed information on its capabilities and use.
|
|
@@ -368,7 +738,7 @@ gKnitting Ruby and R documents quickly.
|
|
|
368
738
|
## The Yaml header
|
|
369
739
|
|
|
370
740
|
An __R markdown__ document should start with a Yaml header and be stored in a file with
|
|
371
|
-
'.Rmd' extension. This document has the following header for
|
|
741
|
+
'.Rmd' extension. This document has the following header for gKnitting an HTML document.
|
|
372
742
|
|
|
373
743
|
```
|
|
374
744
|
---
|
|
@@ -376,7 +746,7 @@ title: "How to do reproducible research in Ruby with gKnit"
|
|
|
376
746
|
author:
|
|
377
747
|
- "Rodrigo Botafogo"
|
|
378
748
|
- "Daniel Mossé - University of Pittsburgh"
|
|
379
|
-
tags: [Tech, Data Science, Ruby, R,
|
|
749
|
+
tags: [Tech, Data Science, Ruby, R, JRuby, Galaaz]
|
|
380
750
|
date: "20/02/2019"
|
|
381
751
|
output:
|
|
382
752
|
html_document:
|
|
@@ -391,6 +761,35 @@ output:
|
|
|
391
761
|
|
|
392
762
|
For more information on the options in the Yaml header, [check here](https://bookdown.org/yihui/rmarkdown/html-document.html).
|
|
393
763
|
|
|
764
|
+
## Choosing the output format when calling gknit
|
|
765
|
+
|
|
766
|
+
Yes: you can select the render target on the **command line**. **`bin/gknit`** (or **`gknit`** on your `PATH`) forwards options to **`rmarkdown::render`** via **`R::Rmarkdown.render`**.
|
|
767
|
+
|
|
768
|
+
* **`--output_format FORMAT`** — name of the format, as in the YAML `output:` block. Examples:
|
|
769
|
+
* **`html_document`** — HTML (often the default you list first under `output:`).
|
|
770
|
+
* **`pdf_document`** — PDF (you need a working LaTeX setup, e.g. TinyTeX; see **`bin/install-tinytex`**).
|
|
771
|
+
* **`md_document`**, **`github_document`**, or any other format defined in your YAML.
|
|
772
|
+
* **`all`** — render **every** format declared under `output:` in the document (same idea as in R Markdown).
|
|
773
|
+
|
|
774
|
+
If you **omit** **`--output_format`**, gknit passes **`NULL`** for the format argument. In that case **rmarkdown** uses the **first** format listed under **`output:`** in the YAML (and if none is specified there, behavior follows the usual rmarkdown defaults, typically HTML).
|
|
775
|
+
|
|
776
|
+
Other useful flags:
|
|
777
|
+
|
|
778
|
+
* **`--output_file NAME`** — output file name (optional path; see also **`--output_dir`**).
|
|
779
|
+
* **`--output_dir DIR`** — directory for the rendered file (created if missing).
|
|
780
|
+
* **`--bridge_timeout_sec`** / **`--callback_timeout_ms`** — longer R or install steps (see elsewhere in this manual).
|
|
781
|
+
|
|
782
|
+
Examples (run from the directory where paths make sense, or use absolute paths):
|
|
783
|
+
|
|
784
|
+
```text
|
|
785
|
+
bin/gknit blogs/manual/manual.Rmd
|
|
786
|
+
bin/gknit --output_format html_document blogs/manual/manual.Rmd
|
|
787
|
+
bin/gknit --output_format pdf_document blogs/manual/manual.Rmd
|
|
788
|
+
bin/gknit --output_format all blogs/manual/manual.Rmd
|
|
789
|
+
```
|
|
790
|
+
|
|
791
|
+
Use **`gknit -h`** for the full option list.
|
|
792
|
+
|
|
394
793
|
## __R Markdown__ formatting
|
|
395
794
|
|
|
396
795
|
Document formatting can be done with simple markups such as:
|
|
@@ -435,7 +834,7 @@ Running and executing Ruby and R code is actually what really interests us is th
|
|
|
435
834
|
Inserting a code chunk is done by adding code in a block delimited by three back ticks
|
|
436
835
|
followed by an open
|
|
437
836
|
curly brace ('{') followed with the engine name (r, ruby, rb, include, ...), an
|
|
438
|
-
any optional chunk_label and options, as shown
|
|
837
|
+
any optional chunk_label and options, as shown below:
|
|
439
838
|
|
|
440
839
|
````
|
|
441
840
|
```{engine_name [chunk_label], [chunk_options]}`r ''`
|
|
@@ -524,7 +923,7 @@ grammar of graphics" [@Wilkinson:grammar_of_graphics]. The idea of the grammar o
|
|
|
524
923
|
is to build a graphics by adding layers to the plot. More information can be found in
|
|
525
924
|
https://towardsdatascience.com/a-comprehensive-guide-to-the-grammar-of-graphics-for-effective-visualization-of-multi-dimensional-1f92b4ed4149.
|
|
526
925
|
|
|
527
|
-
In the plot
|
|
926
|
+
In the plot below the 'mpg' dataset from base R is used. "The data concerns city-cycle fuel
|
|
528
927
|
consumption in miles per gallon, to be predicted in terms of 3 multivalued discrete and 5
|
|
529
928
|
continuous attributes." (Quinlan, 1993)
|
|
530
929
|
|
|
@@ -713,17 +1112,17 @@ Here, for instance, is a table definition in HTML and its output in the document
|
|
|
713
1112
|
</div>
|
|
714
1113
|
|
|
715
1114
|
But manually creating HTML output is not always easy or desirable, specially
|
|
716
|
-
if we intend the document to be rendered in other formats, for example, as
|
|
1115
|
+
if we intend the document to be rendered in other formats, for example, as LaTeX.
|
|
717
1116
|
Also, The above
|
|
718
1117
|
table looks ugly. The 'kableExtra' library is a great library for
|
|
719
1118
|
creating beautiful tables. Take a look at https://cran.r-project.org/web/packages/kableExtra/vignettes/awesome_table_in_html.html
|
|
720
1119
|
|
|
721
1120
|
In the next chunk, we output the 'mtcars' dataframe from R in a nicely formatted
|
|
722
|
-
table. Note that we retrieve the mtcars dataframe by using '
|
|
1121
|
+
table. Note that we retrieve the mtcars dataframe by using '~R[:mtcars]'.
|
|
723
1122
|
|
|
724
1123
|
```{ruby nice_table}
|
|
725
1124
|
R.install_and_loads('kableExtra')
|
|
726
|
-
outputs (
|
|
1125
|
+
outputs (~R[:mtcars]).kable.kable_styling
|
|
727
1126
|
```
|
|
728
1127
|
|
|
729
1128
|
## Including Ruby files in a chunk
|
|
@@ -751,7 +1150,7 @@ true, ruby's 'require\_relative' semantics is used to load the file, when false,
|
|
|
751
1150
|
```
|
|
752
1151
|
````
|
|
753
1152
|
|
|
754
|
-
|
|
1153
|
+
Below we include file 'model.rb', which is in the same directory of this blog.
|
|
755
1154
|
This code uses R 'caret' package to split a dataset in a train and test sets.
|
|
756
1155
|
The 'caret' package is a very important a useful package for doing Data Analysis,
|
|
757
1156
|
it has hundreds of functions for all steps of the Data Analysis workflow. To
|
|
@@ -772,7 +1171,7 @@ will install the package if it is not already installed and can take a while.
|
|
|
772
1171
|
```
|
|
773
1172
|
|
|
774
1173
|
```{ruby model_partition}
|
|
775
|
-
mtcars =
|
|
1174
|
+
mtcars = ~R[:mtcars]
|
|
776
1175
|
model = Model.new(mtcars, percent_train: 0.8)
|
|
777
1176
|
model.partition(:mpg)
|
|
778
1177
|
puts model.train.head
|
|
@@ -784,9 +1183,9 @@ puts model.test.head
|
|
|
784
1183
|
gKnit also allows developers to document and load files that are not in the same directory
|
|
785
1184
|
of the '.Rmd' file.
|
|
786
1185
|
|
|
787
|
-
Here is an example of loading
|
|
788
|
-
is set to FALSE, so Ruby will look for the file in its
|
|
789
|
-
need to
|
|
1186
|
+
Here is an example of loading Ruby’s standard library file `find.rb`. In this example, relative
|
|
1187
|
+
is set to FALSE, so Ruby will look for the file in its `$LOAD_PATH`, and the user does not
|
|
1188
|
+
need to know its directory on disk.
|
|
790
1189
|
|
|
791
1190
|
````
|
|
792
1191
|
```{include find, relative = FALSE}`r ''`
|
|
@@ -808,9 +1207,9 @@ the Yaml header to generate this blog in PDF format instead of HTML:
|
|
|
808
1207
|
|
|
809
1208
|
```
|
|
810
1209
|
---
|
|
811
|
-
title: "gKnit - Ruby and R Knitting with Galaaz
|
|
1210
|
+
title: "gKnit - Ruby and R Knitting with Galaaz"
|
|
812
1211
|
author: "Rodrigo Botafogo"
|
|
813
|
-
tags: [Galaaz, Ruby, R,
|
|
1212
|
+
tags: [Galaaz, Ruby, R, JRuby, knitr, gknit]
|
|
814
1213
|
date: "29 October 2018"
|
|
815
1214
|
output:
|
|
816
1215
|
pdf\_document:
|
|
@@ -822,7 +1221,7 @@ output:
|
|
|
822
1221
|
|
|
823
1222
|
## Template based documents generation
|
|
824
1223
|
|
|
825
|
-
When a document is converted to PDF it follows a certain
|
|
1224
|
+
When a document is converted to PDF it follows a certain conversion template. We've seen above
|
|
826
1225
|
the use of 'galaaz.sty' as a basic template to generate a PDF document. Using the
|
|
827
1226
|
'gknit-draft' app that comes with Galaaz, the same .Rmd file can be compiled to different
|
|
828
1227
|
looking PDF documents. Galaaz automatically loads the 'rticles' R package that comes with
|
|
@@ -872,14 +1271,14 @@ gknit-draft --filename my_r_article --template rjournal_article --package rticle
|
|
|
872
1271
|
|
|
873
1272
|
# Accessing R variables
|
|
874
1273
|
|
|
875
|
-
Galaaz allows Ruby to access variables created in R. For example, the
|
|
876
|
-
available in R and can be accessed from Ruby by using the
|
|
877
|
-
symbol for the variable, in this case
|
|
878
|
-
used to output the
|
|
879
|
-
|
|
1274
|
+
Galaaz allows Ruby to access variables created in R. For example, the `mtcars` data set is
|
|
1275
|
+
available in R and can be accessed from Ruby by using the tilde operator followed by the
|
|
1276
|
+
symbol for the variable, in this case `:mtcars`. In the code below, method `outputs` is
|
|
1277
|
+
used to output the `mtcars` data set nicely formatted in HTML by use of the `kable` and
|
|
1278
|
+
`kable_styling` functions. Method `outputs` is only available when used with gKnit.
|
|
880
1279
|
|
|
881
1280
|
```{ruby view_kable}
|
|
882
|
-
outputs (
|
|
1281
|
+
outputs (~R[:mtcars]).kable.kable_styling
|
|
883
1282
|
```
|
|
884
1283
|
|
|
885
1284
|
# Basic Data Types
|
|
@@ -897,7 +1296,7 @@ table.
|
|
|
897
1296
|
| logical | logical | logical |
|
|
898
1297
|
| integer | numeric | integer |
|
|
899
1298
|
| double | numeric | double |
|
|
900
|
-
| complex | complex |
|
|
1299
|
+
| complex | complex | complex |
|
|
901
1300
|
| character | character | character |
|
|
902
1301
|
| raw | raw | raw |
|
|
903
1302
|
|
|
@@ -974,7 +1373,7 @@ In this next example, method 'c' is chainned after 'vec1'. This also looks like
|
|
|
974
1373
|
method of the vector, but in reallity, this is actually closer to the pipe operator. When
|
|
975
1374
|
Galaaz identifies that 'c' is not a method of 'vec' it actually tries to call 'R.c' with
|
|
976
1375
|
'vec1' as the first argument concatenated with all the other available arguments. The code
|
|
977
|
-
|
|
1376
|
+
below is automatically converted to the code above.
|
|
978
1377
|
|
|
979
1378
|
```{ruby chainning_methods}
|
|
980
1379
|
vec = vec1.c(vec2)
|
|
@@ -1008,7 +1407,7 @@ Vectors can be indexed by using the '[]' operator:
|
|
|
1008
1407
|
puts vec4[3]
|
|
1009
1408
|
```
|
|
1010
1409
|
|
|
1011
|
-
We can also index a vector with another vector. For example, in the code
|
|
1410
|
+
We can also index a vector with another vector. For example, in the code below, we take elements
|
|
1012
1411
|
1, 3, 5, and 7 from vec3:
|
|
1013
1412
|
|
|
1014
1413
|
```{ruby index_by_vector}
|
|
@@ -1179,7 +1578,7 @@ operator) and then the vector was indexed by its first element, extracting the n
|
|
|
1179
1578
|
|
|
1180
1579
|
A data frame is a table like structure in which each column has the same number of
|
|
1181
1580
|
rows. Data frames are the basic structure for storing data for data analysis. We have already
|
|
1182
|
-
seen a data frame previously when we accessed variable '
|
|
1581
|
+
seen a data frame previously when we accessed variable '~R[:mtcars]'. In order to create a
|
|
1183
1582
|
data frame, function 'data__frame' is used:
|
|
1184
1583
|
|
|
1185
1584
|
```{ruby dataframe}
|
|
@@ -1196,45 +1595,45 @@ A data frame can be indexed the same way as a matrix, by using '[row, column]',
|
|
|
1196
1595
|
column can either be a numeric or the name of the row or column
|
|
1197
1596
|
|
|
1198
1597
|
```{ruby dataframe_index}
|
|
1199
|
-
puts (
|
|
1200
|
-
puts (
|
|
1201
|
-
puts (
|
|
1598
|
+
puts (~R[:mtcars]).head
|
|
1599
|
+
puts (~R[:mtcars])[1, 2]
|
|
1600
|
+
puts (~R[:mtcars])['Datsun 710', 'mpg']
|
|
1202
1601
|
```
|
|
1203
1602
|
|
|
1204
1603
|
Extracting a column from a data frame as a vector can be done by using the double square bracket
|
|
1205
1604
|
operator:
|
|
1206
1605
|
|
|
1207
1606
|
```{ruby dataframe_column}
|
|
1208
|
-
puts (
|
|
1607
|
+
puts (~R[:mtcars])[['mpg']]
|
|
1209
1608
|
```
|
|
1210
1609
|
|
|
1211
1610
|
A data frame column can also be accessed as if it were an instance variable of the data frame:
|
|
1212
1611
|
|
|
1213
1612
|
```{ruby dataframe_instance_variable}
|
|
1214
|
-
puts (
|
|
1613
|
+
puts (~R[:mtcars]).mpg
|
|
1215
1614
|
```
|
|
1216
1615
|
|
|
1217
1616
|
Slicing a data frame can be done by indexing it with a vector (we use 'head' to reduce the
|
|
1218
1617
|
output):
|
|
1219
1618
|
|
|
1220
1619
|
```{ruby dataframe_column_slice}
|
|
1221
|
-
puts (
|
|
1620
|
+
puts (~R[:mtcars])[R.c('mpg', 'hp')].head
|
|
1222
1621
|
```
|
|
1223
1622
|
|
|
1224
1623
|
A row slice can be obtained by indexing by row and using the ':all' keyword for the column:
|
|
1225
1624
|
|
|
1226
1625
|
```{ruby dataframe_row_slice}
|
|
1227
|
-
puts (
|
|
1626
|
+
puts (~R[:mtcars])[R.c('Datsun 710', 'Camaro Z28'), :all]
|
|
1228
1627
|
```
|
|
1229
1628
|
|
|
1230
1629
|
Finally, a data frame can also be indexed with a logical vector. In this next example, the
|
|
1231
1630
|
'am' column of :mtcars is compared with 0 (with method 'eq'). When 'am' is equal to 0 the
|
|
1232
|
-
car is automatic. So, by doing '(
|
|
1631
|
+
car is automatic. So, by doing '(~R[:mtcars]).am.eq 0' a logical vector is created with
|
|
1233
1632
|
'true' whenever 'am' is 0 and 'false' otherwise.
|
|
1234
1633
|
|
|
1235
1634
|
```{ruby logical_vector_filter}
|
|
1236
1635
|
# obtain a vector with 'true' for cars with automatic transmission
|
|
1237
|
-
automatic = (
|
|
1636
|
+
automatic = (~R[:mtcars]).am.eq 0
|
|
1238
1637
|
puts automatic
|
|
1239
1638
|
```
|
|
1240
1639
|
|
|
@@ -1243,7 +1642,7 @@ which all cars have automatic transmission.
|
|
|
1243
1642
|
|
|
1244
1643
|
```{ruby dataframe_logical}
|
|
1245
1644
|
# slice the data frame by using this vector
|
|
1246
|
-
puts (
|
|
1645
|
+
puts (~R[:mtcars])[automatic, :all]
|
|
1247
1646
|
```
|
|
1248
1647
|
|
|
1249
1648
|
# Writing Expressions in Galaaz
|
|
@@ -1253,24 +1652,24 @@ Galaaz extends Ruby to work with complex expressions, similar to R's expressions
|
|
|
1253
1652
|
|
|
1254
1653
|
## Expressions from operators
|
|
1255
1654
|
|
|
1256
|
-
The code
|
|
1655
|
+
The code below
|
|
1257
1656
|
creates an expression summing two symbols
|
|
1258
1657
|
|
|
1259
1658
|
```{ruby expressions}
|
|
1260
|
-
exp1 = :a + :b
|
|
1659
|
+
exp1 = R[:a] + R[:b]
|
|
1261
1660
|
puts exp1
|
|
1262
1661
|
```
|
|
1263
1662
|
We can build any complex mathematical expression
|
|
1264
1663
|
|
|
1265
1664
|
```{ruby expr2}
|
|
1266
|
-
exp2 = (:a + :b) * 2.0 + :c ** 2 / :z
|
|
1665
|
+
exp2 = (R[:a] + R[:b]) * 2.0 + R[:c] ** 2 / R[:z]
|
|
1267
1666
|
puts exp2
|
|
1268
1667
|
```
|
|
1269
1668
|
|
|
1270
1669
|
It is also possible to use inequality operators in building expressions
|
|
1271
1670
|
|
|
1272
1671
|
```{ruby expr3}
|
|
1273
|
-
exp3 = (:a + :b) >= :z
|
|
1672
|
+
exp3 = (R[:a] + R[:b]) >= :z
|
|
1274
1673
|
puts exp3
|
|
1275
1674
|
```
|
|
1276
1675
|
|
|
@@ -1279,7 +1678,7 @@ notation for those operators such as (.gt, .ge, etc.). So the same expression w
|
|
|
1279
1678
|
above can also be written as
|
|
1280
1679
|
|
|
1281
1680
|
```{ruby expr4}
|
|
1282
|
-
exp4 = (:a + :b).ge :z
|
|
1681
|
+
exp4 = (R[:a] + R[:b]).ge :z
|
|
1283
1682
|
puts exp4
|
|
1284
1683
|
```
|
|
1285
1684
|
|
|
@@ -1288,23 +1687,23 @@ those are expressions involving '==', and '='. In order to write an expression
|
|
|
1288
1687
|
need to use the method '.eq' and for '=' we need the function '.assign'
|
|
1289
1688
|
|
|
1290
1689
|
```{ruby expr5}
|
|
1291
|
-
exp5 = (:a + :b).eq :z
|
|
1690
|
+
exp5 = (R[:a] + R[:b]).eq :z
|
|
1292
1691
|
puts exp5
|
|
1293
1692
|
```
|
|
1294
1693
|
|
|
1295
1694
|
```{ruby expr6}
|
|
1296
|
-
exp6 = :y.assign :a + :b
|
|
1695
|
+
exp6 = R[:y].assign R[:a] + R[:b]
|
|
1297
1696
|
puts exp6
|
|
1298
1697
|
```
|
|
1299
1698
|
In general we think that using the functional notation is preferable to using the
|
|
1300
1699
|
symbolic notation as otherwise, we end up writing invalid expressions such as
|
|
1301
1700
|
|
|
1302
1701
|
```{ruby exp_wrong, warning=FALSE, eval=FALSE}
|
|
1303
|
-
exp_wrong = (:a + :b) == :z
|
|
1702
|
+
exp_wrong = (R[:a] + R[:b]) == :z
|
|
1304
1703
|
puts exp_wrong
|
|
1305
1704
|
```
|
|
1306
1705
|
and it might be difficult to understand what is going on here. The problem lies with the fact that
|
|
1307
|
-
when using '==' we are comparing expression (:a + :b) to expression :z with '=='. When the
|
|
1706
|
+
when using '==' we are comparing expression (R[:a] + R[:b]) to expression :z with '=='. When the
|
|
1308
1707
|
comparison is executed, the system tries to evaluate :a, :b and :z, and those symbols at
|
|
1309
1708
|
this time are not bound to anything and we get a "object 'a' not found" message.
|
|
1310
1709
|
If we only use functional notation, this type of error will not occur.
|
|
@@ -1319,21 +1718,21 @@ When we want the function to be part of the expression, we call the function pre
|
|
|
1319
1718
|
by the letter E, such as 'E.sin(x)'
|
|
1320
1719
|
|
|
1321
1720
|
```{ruby method_expression}
|
|
1322
|
-
exp7 = :y.assign E.sin(:x)
|
|
1721
|
+
exp7 = R[:y].assign E.sin(R[:x])
|
|
1323
1722
|
puts exp7
|
|
1324
1723
|
```
|
|
1325
1724
|
|
|
1326
1725
|
Expressions can also be written using '.' notation:
|
|
1327
1726
|
|
|
1328
1727
|
```{ruby expression_with_dot}
|
|
1329
|
-
exp8 = :y.assign :x.sin
|
|
1728
|
+
exp8 = R[:y].assign R[:x].sin
|
|
1330
1729
|
puts exp8
|
|
1331
1730
|
```
|
|
1332
1731
|
|
|
1333
1732
|
When a function has multiple arguments, the first one can be used before the '.':
|
|
1334
1733
|
|
|
1335
1734
|
```{ruby expression_multiple_args}
|
|
1336
|
-
exp9 = :x.c(:y)
|
|
1735
|
+
exp9 = R[:x].c(R[:y])
|
|
1337
1736
|
puts exp9
|
|
1338
1737
|
```
|
|
1339
1738
|
|
|
@@ -1343,7 +1742,7 @@ Expressions can be evaluated by calling function 'eval' with a binding. A bindin
|
|
|
1343
1742
|
with a list:
|
|
1344
1743
|
|
|
1345
1744
|
```{ruby eval_expression_list}
|
|
1346
|
-
exp = (:a + :b) * 2.0 + :c ** 2 / :z
|
|
1745
|
+
exp = (R[:a] + R[:b]) * 2.0 + R[:c] ** 2 / R[:z]
|
|
1347
1746
|
puts exp.eval(R.list(a: 10, b: 20, c: 30, z: 40))
|
|
1348
1747
|
```
|
|
1349
1748
|
|
|
@@ -1362,7 +1761,7 @@ puts exp.eval(df)
|
|
|
1362
1761
|
# Manipulating Data
|
|
1363
1762
|
|
|
1364
1763
|
One of the major benefits of Galaaz is to bring strong data manipulation to Ruby. The following
|
|
1365
|
-
examples were extracted from
|
|
1764
|
+
examples were extracted from Hadley's "R for Data Science" (https://r4ds.had.co.nz/). This
|
|
1366
1765
|
is a highly recommended book for those not already familiar with the 'tidyverse' style of
|
|
1367
1766
|
programming in R. In the sections to follow, we will limit ourselves to convert the R code to
|
|
1368
1767
|
Galaaz.
|
|
@@ -1374,9 +1773,9 @@ locally, and if not, installs it. This data frame contains all 336,776 flights t
|
|
|
1374
1773
|
departed from New York City in 2013. The data comes from the US Bureau of
|
|
1375
1774
|
Transportation Statistics.
|
|
1376
1775
|
|
|
1377
|
-
Dplyr uses
|
|
1378
|
-
|
|
1379
|
-
|
|
1776
|
+
Dplyr often uses **tibbles** in place of classic data frames. In Galaaz, printing may differ from
|
|
1777
|
+
the R console; if you need a classic tabular printout, convert with **`as__data__frame`** (or use
|
|
1778
|
+
`head` / `str` in R via `R` calls).
|
|
1380
1779
|
|
|
1381
1780
|
```{ruby nycflights13}
|
|
1382
1781
|
R.install_and_loads('nycflights13')
|
|
@@ -1384,17 +1783,17 @@ R.library('dplyr')
|
|
|
1384
1783
|
```
|
|
1385
1784
|
|
|
1386
1785
|
```{ruby flights}
|
|
1387
|
-
flights =
|
|
1786
|
+
flights = ~R[:flights]
|
|
1388
1787
|
puts flights.head
|
|
1389
1788
|
```
|
|
1390
1789
|
|
|
1391
1790
|
## Filtering rows with Filter
|
|
1392
1791
|
|
|
1393
1792
|
In this example we filter the flights data set by giving to the filter function two expressions:
|
|
1394
|
-
the first :month.eq 1
|
|
1793
|
+
the first R[:month].eq 1
|
|
1395
1794
|
|
|
1396
1795
|
```{ruby filter_rows}
|
|
1397
|
-
puts flights.filter((:month.eq 1), (:day.eq 1)).head
|
|
1796
|
+
puts flights.filter((R[:month].eq 1), (R[:day].eq 1)).head
|
|
1398
1797
|
```
|
|
1399
1798
|
|
|
1400
1799
|
## Logical Operators
|
|
@@ -1402,7 +1801,7 @@ puts flights.filter((:month.eq 1), (:day.eq 1)).head
|
|
|
1402
1801
|
All flights that departed in November of December
|
|
1403
1802
|
|
|
1404
1803
|
```{ruby nov_dec}
|
|
1405
|
-
puts flights.filter((:month.eq 11) | (:month.eq 12)).head
|
|
1804
|
+
puts flights.filter((R[:month].eq 11) | (R[:month].eq 12)).head
|
|
1406
1805
|
```
|
|
1407
1806
|
|
|
1408
1807
|
The same as above, but using the 'in' operator. In R, it is possible to define many operators
|
|
@@ -1411,7 +1810,7 @@ operators from Galaaz the '._' method is used, where the first argument is the o
|
|
|
1411
1810
|
symbol, in this case ':in' and the second argument is the vector:
|
|
1412
1811
|
|
|
1413
1812
|
```{ruby in_op}
|
|
1414
|
-
puts flights.filter(:month._ :in, R.c(11, 12)).head
|
|
1813
|
+
puts flights.filter(R[:month]._ :in, R.c(11, 12)).head
|
|
1415
1814
|
```
|
|
1416
1815
|
|
|
1417
1816
|
## Filtering with NA (Not Available)
|
|
@@ -1430,13 +1829,13 @@ Now filtering by :x > 1 shows all lines that satisfy this condition, where the r
|
|
|
1430
1829
|
not.
|
|
1431
1830
|
|
|
1432
1831
|
```{ruby filter_na}
|
|
1433
|
-
puts df.filter(:x > 1)
|
|
1832
|
+
puts df.filter(R[:x] > 1)
|
|
1434
1833
|
```
|
|
1435
1834
|
|
|
1436
1835
|
To match an NA use method 'is__na'
|
|
1437
1836
|
|
|
1438
1837
|
```{ruby with_na}
|
|
1439
|
-
puts df.filter((:x.is__na) | (:x > 1))
|
|
1838
|
+
puts df.filter((R[:x].is__na) | (R[:x] > 1))
|
|
1440
1839
|
```
|
|
1441
1840
|
|
|
1442
1841
|
## Arrange Rows with arrange
|
|
@@ -1450,7 +1849,7 @@ puts flights.arrange(:year, :month, :day).head
|
|
|
1450
1849
|
To arrange in descending order, use function 'desc'
|
|
1451
1850
|
|
|
1452
1851
|
```{ruby desc_arrange}
|
|
1453
|
-
puts flights.arrange(:dep_delay.desc).head
|
|
1852
|
+
puts flights.arrange(R[:dep_delay].desc).head
|
|
1454
1853
|
```
|
|
1455
1854
|
|
|
1456
1855
|
## Selecting columns
|
|
@@ -1464,7 +1863,7 @@ puts flights.select(:year, :month, :day).head
|
|
|
1464
1863
|
It is also possible to select column in a given range
|
|
1465
1864
|
|
|
1466
1865
|
```{ruby select_range}
|
|
1467
|
-
puts flights.select(:year.up_to
|
|
1866
|
+
puts flights.select(R[:year].up_to(R[:day])).head
|
|
1468
1867
|
```
|
|
1469
1868
|
|
|
1470
1869
|
Select all columns that start with a given name sequence
|
|
@@ -1494,7 +1893,7 @@ puts flights.select(:year, :month, :day, E.everything).head
|
|
|
1494
1893
|
|
|
1495
1894
|
```{ruby small_flights}
|
|
1496
1895
|
flights_sm = flights.
|
|
1497
|
-
select((:year.up_to
|
|
1896
|
+
select((R[:year].up_to(R[:day])),
|
|
1498
1897
|
E.ends_with('delay'),
|
|
1499
1898
|
:distance,
|
|
1500
1899
|
:air_time)
|
|
@@ -1504,8 +1903,8 @@ puts flights_sm.head
|
|
|
1504
1903
|
|
|
1505
1904
|
```{ruby mutate}
|
|
1506
1905
|
flights_sm = flights_sm.
|
|
1507
|
-
mutate(gain: :dep_delay - :arr_delay,
|
|
1508
|
-
speed: :distance / :air_time * 60)
|
|
1906
|
+
mutate(gain: R[:dep_delay] - R[:arr_delay],
|
|
1907
|
+
speed: R[:distance] / R[:air_time] * 60)
|
|
1509
1908
|
puts flights_sm.head
|
|
1510
1909
|
```
|
|
1511
1910
|
|
|
@@ -1522,7 +1921,7 @@ When a data frame is grouped with 'group_by' summaries apply to the given group:
|
|
|
1522
1921
|
|
|
1523
1922
|
```{ruby summarise_group_by}
|
|
1524
1923
|
by_day = flights.group_by(:year, :month, :day)
|
|
1525
|
-
puts by_day.summarise(delay: :dep_delay.mean(na__rm: true)).head
|
|
1924
|
+
puts by_day.summarise(delay: R[:dep_delay].mean(na__rm: true)).head
|
|
1526
1925
|
```
|
|
1527
1926
|
|
|
1528
1927
|
Next we put many operations together by pipping them one after the other:
|
|
@@ -1532,23 +1931,25 @@ delays = flights.
|
|
|
1532
1931
|
group_by(:dest).
|
|
1533
1932
|
summarise(
|
|
1534
1933
|
count: E.n,
|
|
1535
|
-
dist: :distance.mean(na__rm: true),
|
|
1536
|
-
delay: :arr_delay.mean(na__rm: true)).
|
|
1537
|
-
filter(:count > 20, :dest != "NHL")
|
|
1934
|
+
dist: R[:distance].mean(na__rm: true),
|
|
1935
|
+
delay: R[:arr_delay].mean(na__rm: true)).
|
|
1936
|
+
filter(R[:count] > 20, R[:dest] != "NHL")
|
|
1538
1937
|
|
|
1539
1938
|
puts delays.head
|
|
1540
1939
|
```
|
|
1541
1940
|
|
|
1542
1941
|
# Using Data Table
|
|
1543
1942
|
|
|
1943
|
+
The next chunk converts the **nycflights13** `flights` tibble already loaded above into a
|
|
1944
|
+
**`data.table`**. That keeps the manual offline and avoids downloading a remote CSV during gknit
|
|
1945
|
+
(network stalls look like bridge hangs when the transfer runs inside a single R eval).
|
|
1946
|
+
|
|
1544
1947
|
```{ruby fread}
|
|
1545
1948
|
R.library('data.table')
|
|
1546
|
-
R.install_and_loads('curl')
|
|
1547
1949
|
|
|
1548
|
-
|
|
1549
|
-
flights = R.fread(input)
|
|
1550
|
-
puts flights
|
|
1950
|
+
flights = R.as__data__table(~R[:flights])
|
|
1551
1951
|
puts flights.dim
|
|
1952
|
+
puts R.head(flights, 12)
|
|
1552
1953
|
```
|
|
1553
1954
|
|
|
1554
1955
|
```{ruby data_table}
|
|
@@ -1566,7 +1967,7 @@ puts data_table.ID
|
|
|
1566
1967
|
|
|
1567
1968
|
```{ruby subset_i}
|
|
1568
1969
|
# subset rows in i
|
|
1569
|
-
ans = flights[(:origin.eq "JFK") & (:month.eq 6)]
|
|
1970
|
+
ans = flights[(R[:origin].eq "JFK") & (R[:month].eq 6)]
|
|
1570
1971
|
puts ans.head
|
|
1571
1972
|
|
|
1572
1973
|
# Get the first two rows from flights.
|
|
@@ -1574,8 +1975,7 @@ puts ans.head
|
|
|
1574
1975
|
ans = flights[(1..2)]
|
|
1575
1976
|
puts ans
|
|
1576
1977
|
|
|
1577
|
-
# Sort
|
|
1578
|
-
|
|
1978
|
+
# Sort by origin asc, then dest desc (example kept commented):
|
|
1579
1979
|
# ans = flights[E.order(:origin, -(:dest))]
|
|
1580
1980
|
# puts ans.head
|
|
1581
1981
|
|
|
@@ -1588,66 +1988,274 @@ puts ans
|
|
|
1588
1988
|
ans = flights[:all, :arr_delay]
|
|
1589
1989
|
puts ans.head
|
|
1590
1990
|
|
|
1591
|
-
#
|
|
1991
|
+
# arr_delay as data.table (not plain vector).
|
|
1592
1992
|
|
|
1593
|
-
ans = flights[:all, :arr_delay.list]
|
|
1993
|
+
ans = flights[:all, R[:arr_delay].list]
|
|
1594
1994
|
puts ans.head
|
|
1595
1995
|
|
|
1596
|
-
ans = flights[:all, E.list(:arr_delay, :dep_delay)]
|
|
1996
|
+
ans = flights[:all, E.list(R[:arr_delay], R[:dep_delay])]
|
|
1997
|
+
```
|
|
1998
|
+
|
|
1999
|
+
# Apache Arrow
|
|
2000
|
+
|
|
2001
|
+
[Apache Arrow](https://arrow.apache.org/) is a **columnar** in-memory format used heavily in R
|
|
2002
|
+
and Python for analytics. In Galaaz, **Ruby does not hold an Arrow C++ table itself**; instead you
|
|
2003
|
+
build ordinary Ruby structures (arrays of row hashes), and **`R::Arrow.from_ruby_batches`** creates
|
|
2004
|
+
a real **Arrow `Table` inside GNU R**. From there you use R’s **`arrow`** and **`dplyr`** packages
|
|
2005
|
+
as usual: **`group_by`** on the Arrow table, **`summarise`** for aggregates, then **`collect()`** to
|
|
2006
|
+
materialize a tibble when you need in-memory R rows.
|
|
2007
|
+
|
|
2008
|
+
That pattern matches production use: **JRuby threads** (or sequential code) assemble many rows in
|
|
2009
|
+
Ruby; you pay **one** bridge-heavy handoff to R; **dplyr** runs vectorised work on the Arrow table
|
|
2010
|
+
in R.
|
|
2011
|
+
|
|
2012
|
+
**Prerequisites:** install R packages **`arrow`** and **`dplyr`**. Run scripts with
|
|
2013
|
+
**`bin/galaaz-jruby`** (or the same JVM flags as in **`docs/testing.md`**) so the Arrow JNI stack is
|
|
2014
|
+
available.
|
|
2015
|
+
|
|
2016
|
+
## Other `R::Arrow` helpers
|
|
2017
|
+
|
|
2018
|
+
The Ruby module **`R::Arrow`** (see `lib/R_interface/r_arrow.rb`) also includes:
|
|
2019
|
+
|
|
2020
|
+
* **`R::Arrow.table_from(df)`** — wrap an R `data.frame` / tibble as an Arrow table.
|
|
2021
|
+
* **`R::Arrow.read_feather` / `write_feather`**, **`read_parquet`**, **`dataset(path)`** — file and
|
|
2022
|
+
dataset IO on paths visible to R.
|
|
2023
|
+
|
|
2024
|
+
## Example: many Ruby rows → Arrow in R → grouped statistics
|
|
2025
|
+
|
|
2026
|
+
The repository test **`slow-specs/arrow_large_pipeline_spec.rb`** builds **200k rows** in parallel
|
|
2027
|
+
(eight threads × 25,000 rows), pushes them through **`R::Arrow.from_ruby_batches`**, then checks that
|
|
2028
|
+
**dplyr** group summaries match a Ruby reference calculation. The same logic appears below at a
|
|
2029
|
+
**smaller scale** so this manual can knit quickly; increase `thread_count` and `rows_per_thread`
|
|
2030
|
+
when experimenting locally.
|
|
2031
|
+
|
|
2032
|
+
```{ruby arrow_pipeline_example, message=FALSE, warning=FALSE}
|
|
2033
|
+
# Scaled-down version of slow-specs/arrow_large_pipeline_spec.rb.
|
|
2034
|
+
unless R::Support.eval("requireNamespace('arrow', quietly=TRUE) && requireNamespace('dplyr', quietly=TRUE)") == true
|
|
2035
|
+
puts '(Skip: need arrow + dplyr in R; use bin/galaaz-jruby outside gKnit.)'
|
|
2036
|
+
else
|
|
2037
|
+
thread_count = 4
|
|
2038
|
+
rows_per_thread = 500
|
|
2039
|
+
group_count = 5
|
|
2040
|
+
|
|
2041
|
+
batches = []
|
|
2042
|
+
mutex = Mutex.new
|
|
2043
|
+
threads = []
|
|
2044
|
+
|
|
2045
|
+
thread_count.times do |tid|
|
|
2046
|
+
threads << Thread.new do
|
|
2047
|
+
start = tid * rows_per_thread
|
|
2048
|
+
local = (start...(start + rows_per_thread)).map do |i|
|
|
2049
|
+
{
|
|
2050
|
+
id: i,
|
|
2051
|
+
grp: "g#{i % group_count}",
|
|
2052
|
+
value: (i % 17) + 1,
|
|
2053
|
+
weight: ((i % 5) + 1) * 0.5
|
|
2054
|
+
}
|
|
2055
|
+
end
|
|
2056
|
+
mutex.synchronize { batches << local }
|
|
2057
|
+
end
|
|
2058
|
+
end
|
|
2059
|
+
threads.each(&:join)
|
|
2060
|
+
|
|
2061
|
+
tbl = R::Arrow.from_ruby_batches(batches)
|
|
2062
|
+
puts "R class after from_ruby_batches: #{tbl.rclass}"
|
|
2063
|
+
|
|
2064
|
+
grouped = R.dplyr___group_by(tbl, :grp)
|
|
2065
|
+
summarised = R.dplyr___summarise(
|
|
2066
|
+
grouped,
|
|
2067
|
+
n: E.n(),
|
|
2068
|
+
total: E.sum(:value),
|
|
2069
|
+
wsum: E.sum(R[:value] * R[:weight])
|
|
2070
|
+
)
|
|
2071
|
+
out = R.dplyr___collect(summarised)
|
|
2072
|
+
|
|
2073
|
+
puts 'Per-group summary (first rows):'
|
|
2074
|
+
puts R.as__data__frame(out).head(10)
|
|
2075
|
+
|
|
2076
|
+
total_n = 0
|
|
2077
|
+
(1..(out.nrow >> 0)).each { |i| total_n += (out[['n']][i] >> 0) }
|
|
2078
|
+
puts "Sum of group counts n (should equal #{thread_count * rows_per_thread}): #{total_n}"
|
|
2079
|
+
end
|
|
2080
|
+
```
|
|
2081
|
+
|
|
2082
|
+
**What to notice:** (1) Ruby only sees **`Hash`** rows and Ruby **`Thread`** objects; (2) a single
|
|
2083
|
+
**`from_ruby_batches`** call creates the Arrow table in R; (3) **`dplyr___group_by`** /
|
|
2084
|
+
**`dplyr___summarise`** / **`dplyr___collect`** mirror **`dplyr::group_by`** /
|
|
2085
|
+
**`dplyr::summarise`** / **`dplyr::collect`** on an Arrow-backed table. For a lighter test, see
|
|
2086
|
+
**`specs/arrow_from_ruby_batches_spec.rb`**; for the full-size benchmark, run
|
|
2087
|
+
**`bin/run_slow_rspec slow-specs/arrow_large_pipeline_spec.rb`**.
|
|
2088
|
+
|
|
2089
|
+
# Bioconductor and DESeq2
|
|
2090
|
+
|
|
2091
|
+
**Bioconductor** packages are ordinary R packages installed from the Bioconductor repositories.
|
|
2092
|
+
Galaaz does not treat them specially: once installed in **GNU R**, you load them with
|
|
2093
|
+
**`R.library`** like any CRAN package.
|
|
2094
|
+
|
|
2095
|
+
## Installing Bioconductor packages
|
|
2096
|
+
|
|
2097
|
+
From an R session (or `R -e '...'`), use **BiocManager** (see
|
|
2098
|
+
[bioconductor.org](https://bioconductor.org/install/)):
|
|
2099
|
+
|
|
2100
|
+
```r
|
|
2101
|
+
if (!requireNamespace("BiocManager", quietly = TRUE))
|
|
2102
|
+
install.packages("BiocManager")
|
|
2103
|
+
BiocManager::install(c("DESeq2", "airway"))
|
|
2104
|
+
```
|
|
2105
|
+
|
|
2106
|
+
The **`airway`** package ships the example **`SummarizedExperiment`** used below. **DESeq2**
|
|
2107
|
+
pulls in several dependencies; the first install can take several minutes.
|
|
2108
|
+
|
|
2109
|
+
## Example: DESeq2 on the airway dataset
|
|
2110
|
+
|
|
2111
|
+
The script **`examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb`** is the canonical
|
|
2112
|
+
version in the repository. Run it from the **Galaaz repository root** with JRuby, for example:
|
|
2113
|
+
|
|
2114
|
+
```text
|
|
2115
|
+
bin/galaaz-jruby examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb
|
|
2116
|
+
```
|
|
2117
|
+
|
|
2118
|
+
The workflow in Ruby mirrors a standard DESeq2 vignette:
|
|
2119
|
+
|
|
2120
|
+
1. **`R.library('DESeq2')`** and **`R.library('airway')`**, then **`R.data('airway')`** so the
|
|
2121
|
+
object exists in R’s global environment.
|
|
2122
|
+
2. **`airway = ~R[:airway]`** pulls the experiment into a Galaaz wrapper so you can pass it to R
|
|
2123
|
+
functions as a Ruby value.
|
|
2124
|
+
3. **`R.DESeqDataSet(..., design: (R[:all].til R[:cell] + R[:dex]))`** builds the **`DESeqDataSet`**. The
|
|
2125
|
+
**`(R[:all].til R[:cell] + R[:dex])`** form is Galaaz’s way of passing the one-sided formula
|
|
2126
|
+
**`~ cell + dex`** (adjust for the design you need).
|
|
2127
|
+
4. Prefilter rows with almost no counts: **`keep = R.rowSums(R.counts(dds)) >= 10`** and
|
|
2128
|
+
**`dds = dds[keep, :all]`**.
|
|
2129
|
+
5. **`dds = R.DESeq(dds)`** fits the model; **`res = R.results(dds, contrast: R.c('dex', 'trt', 'untrt'))`**
|
|
2130
|
+
extracts the treatment contrast (adjust **`contrast`** for your experiment).
|
|
2131
|
+
6. Summaries use normal Ruby string interpolation on **`R.nrow`**, **`R.ncol`**, **`R.colnames`**, etc.
|
|
2132
|
+
7. **`R.pdf(...); R.plotMA(res, ...); R.dev__off`** writes DESeq2’s MA plot (path is relative to the
|
|
2133
|
+
process working directory—use the repo root when running the bundled script).
|
|
2134
|
+
|
|
2135
|
+
Related benchmarks and warm-run notes live under **`docs/deseq2_airway_benchmark.md`** and
|
|
2136
|
+
**`examples/bioconductor_deseq2_airway/bench_*.rb`**.
|
|
2137
|
+
|
|
2138
|
+
Below is the full listing (same as the file in the repository). It is **not** executed while this
|
|
2139
|
+
manual is knitted, because **DESeq2** is heavy and may be absent on the build machine.
|
|
2140
|
+
|
|
2141
|
+
```{ruby deseq2_airway_full_listing, eval=FALSE}
|
|
2142
|
+
# Canonical script: examples/bioconductor_deseq2_airway/deseq2_airway_galaaz.rb
|
|
2143
|
+
# Run: bin/galaaz-jruby examples/.../deseq2_airway_galaaz.rb (repo root).
|
|
2144
|
+
|
|
2145
|
+
require 'galaaz'
|
|
2146
|
+
|
|
2147
|
+
R.library('DESeq2')
|
|
2148
|
+
R.library('airway')
|
|
2149
|
+
R.data('airway')
|
|
2150
|
+
|
|
2151
|
+
airway = ~R[:airway]
|
|
2152
|
+
|
|
2153
|
+
# Build DESeq2 dataset with one-sided formula: ~ cell + dex.
|
|
2154
|
+
dds = R.DESeqDataSet(airway, design: (R[:all].til R[:cell] + R[:dex]))
|
|
2155
|
+
|
|
2156
|
+
# Prefilter genes with almost no counts.
|
|
2157
|
+
keep = R.rowSums(R.counts(dds)) >= 10
|
|
2158
|
+
dds = dds[keep, :all]
|
|
2159
|
+
|
|
2160
|
+
# Fit DE model and extract treatment effect.
|
|
2161
|
+
dds = R.DESeq(dds)
|
|
2162
|
+
res = R.results(dds, contrast: R.c('dex', 'trt', 'untrt'))
|
|
2163
|
+
|
|
2164
|
+
# Compact sanity outputs for quick verification.
|
|
2165
|
+
puts "Samples: #{R.ncol(dds)}"
|
|
2166
|
+
puts "Genes after prefilter: #{R.nrow(dds)}"
|
|
2167
|
+
puts "Result rows: #{R.nrow(res)}"
|
|
2168
|
+
puts "Result columns: #{R.colnames(res)}"
|
|
2169
|
+
puts "Significant genes (padj < 0.05): #{R.sum(res.padj < 0.05, na__rm: true)}"
|
|
2170
|
+
|
|
2171
|
+
res_ordered = res[R.order(res.padj), :all]
|
|
2172
|
+
puts R.head(R.as__data__frame(res_ordered), 10)
|
|
2173
|
+
|
|
2174
|
+
# Standard DESeq2 plot call written to file.
|
|
2175
|
+
R.pdf('examples/bioconductor_deseq2_airway/plotMA_galaaz.pdf')
|
|
2176
|
+
R.plotMA(res, ylim: R.c(-5, 5))
|
|
2177
|
+
R.dev__off
|
|
1597
2178
|
```
|
|
1598
2179
|
|
|
2180
|
+
If **DESeq2** and **airway** are installed, the next chunk loads the data and prints a short
|
|
2181
|
+
preview (it does **not** run **`DESeq`** so the manual knits quickly).
|
|
2182
|
+
|
|
2183
|
+
```{ruby deseq2_airway_smoke, message=FALSE, warning=FALSE}
|
|
2184
|
+
unless R::Support.eval("requireNamespace('DESeq2', quietly=TRUE) && requireNamespace('airway', quietly=TRUE)")
|
|
2185
|
+
puts '(Skip: install DESeq2 and airway via BiocManager in R to run the full example.)'
|
|
2186
|
+
else
|
|
2187
|
+
R.library('DESeq2')
|
|
2188
|
+
R.library('airway')
|
|
2189
|
+
R.data('airway')
|
|
2190
|
+
airway = ~R[:airway]
|
|
2191
|
+
puts 'airway object (head of assay / dims via R):'
|
|
2192
|
+
puts "ncol(samples): #{R.ncol(airway)}"
|
|
2193
|
+
puts R.head(R.assay(airway), 3)
|
|
2194
|
+
end
|
|
2195
|
+
```
|
|
2196
|
+
|
|
2197
|
+
# Performance
|
|
2198
|
+
|
|
2199
|
+
For realistic analyses, **most wall-clock time is spent inside GNU R** (model fitting, I/O inside
|
|
2200
|
+
R, graphics). The Galaaz **bridge** adds overhead mainly from **starting a session**, **serializing
|
|
2201
|
+
requests**, and **wrapping results** in Ruby objects—not from reimplementing R’s numerical work.
|
|
2202
|
+
|
|
2203
|
+
Practical tips:
|
|
2204
|
+
|
|
2205
|
+
* Keep **hot loops** in R or vectorized code when possible; use Ruby for orchestration, I/O, and
|
|
2206
|
+
glue.
|
|
2207
|
+
* **Reuse one process**: running many short scripts cold-starts Ruby, the JVM, and R each time;
|
|
2208
|
+
a long-lived process or repeated calls in one run amortize setup (see benchmarks below).
|
|
2209
|
+
* **Batch data**: merge shards in Ruby, then call **`R::Arrow.from_ruby_batches`** (or build one
|
|
2210
|
+
data frame) instead of millions of tiny R calls.
|
|
2211
|
+
|
|
2212
|
+
For measured discussion (including DESeq2-style workloads and warm comparisons), see
|
|
2213
|
+
**`docs/performance.md`** and **`docs/deseq2_airway_benchmark.md`** in the Galaaz repository.
|
|
2214
|
+
|
|
1599
2215
|
# Graphics in Galaaz
|
|
1600
2216
|
|
|
1601
2217
|
Creating graphics in Galaaz is quite easy, as it can use all the power of ggplot2. There are
|
|
1602
|
-
many resources
|
|
2218
|
+
many resources on the web that teach ggplot, so here we give a quick example of ggplot
|
|
1603
2219
|
integration with Ruby. We continue to use the :mtcars dataset and we will plot a diverging
|
|
1604
|
-
bar plot, showing cars that have 'above' or 'below' gas
|
|
2220
|
+
bar plot, showing cars that have 'above' or 'below' gas consumption. Let's first prepare
|
|
1605
2221
|
the data frame with the necessary data:
|
|
1606
2222
|
|
|
1607
2223
|
```{ruby diverging_plot_pre}
|
|
1608
|
-
#
|
|
1609
|
-
mtcars =
|
|
1610
|
-
|
|
1611
|
-
#
|
|
1612
|
-
|
|
1613
|
-
|
|
1614
|
-
mtcars.
|
|
1615
|
-
|
|
1616
|
-
# compute normalized mpg and add it to a new column called mpg_z
|
|
1617
|
-
# Note that the mean value for mpg can be obtained by calling the 'mean'
|
|
1618
|
-
# function on the vector 'mtcars.mpg'. The same with the standard
|
|
1619
|
-
# deviation 'sd'. The vector is then rounded to two digits with 'round 2'
|
|
2224
|
+
# :mtcars -> Ruby handle
|
|
2225
|
+
mtcars = ~R[:mtcars]
|
|
2226
|
+
|
|
2227
|
+
# Row labels are not a plot column; copy them to car_name.
|
|
2228
|
+
mtcars.car_name = R.rownames(R[:mtcars])
|
|
2229
|
+
|
|
2230
|
+
# Z-score mpg (mean/sd on mtcars.mpg); round to 2 decimals.
|
|
1620
2231
|
mtcars.mpg_z = ((mtcars.mpg - mtcars.mpg.mean)/mtcars.mpg.sd).round 2
|
|
1621
2232
|
|
|
1622
|
-
#
|
|
1623
|
-
# that looks at every element of the mpg_z vector and if the value is below
|
|
1624
|
-
# 0, returns 'below', otherwise returns 'above'
|
|
2233
|
+
# ifelse is vectorized: below / above average mpg_z.
|
|
1625
2234
|
mtcars.mpg_type = (mtcars.mpg_z < 0).ifelse("below", "above")
|
|
1626
2235
|
|
|
1627
|
-
#
|
|
2236
|
+
# Sort rows by mpg_z.
|
|
1628
2237
|
mtcars = mtcars[mtcars.mpg_z.order, :all]
|
|
1629
2238
|
|
|
1630
|
-
#
|
|
2239
|
+
# Factor car_name so plot order follows sort.
|
|
1631
2240
|
mtcars.car_name = mtcars.car_name.factor levels: mtcars.car_name
|
|
1632
2241
|
|
|
1633
|
-
# let's look at the final data frame
|
|
1634
2242
|
puts mtcars.head
|
|
1635
2243
|
```
|
|
1636
|
-
Now,
|
|
1637
|
-
|
|
2244
|
+
Now, let's plot the diverging bar plot. When using gKnit, you normally do **not** need to open a
|
|
2245
|
+
graphics device manually; gKnit arranges the figure device for chunk output. Galaaz
|
|
1638
2246
|
provides integration with ggplot. The interested reader should check online for more
|
|
1639
2247
|
information on ggplot, since it is outside the scope of this manual describing
|
|
1640
|
-
how ggplot works.
|
|
2248
|
+
how ggplot works. Here we give only a brief description of how this plot is generated.
|
|
1641
2249
|
|
|
1642
|
-
ggplot implements the 'grammar of graphics'. In this approach, plots are
|
|
2250
|
+
ggplot implements the 'grammar of graphics'. In this approach, plots are built by
|
|
1643
2251
|
adding layers to the plot. On the first layer we describe what we want on the 'x'
|
|
1644
2252
|
and 'y' axis of the plot. In this case, we have 'car_name' on the 'x' axis and
|
|
1645
2253
|
'mpg\_z' on the 'y' axis. Then the type of graph is specified by adding
|
|
1646
2254
|
'geom\_bar' (for a bar graph). We specify that our bars should be filled using
|
|
1647
|
-
'mpg\_type', which is either 'above' or '
|
|
2255
|
+
'mpg\_type', which is either 'above' or 'below' giving then two colours for
|
|
1648
2256
|
filling. On the next layer we specify the labels for the graph, then we add the
|
|
1649
2257
|
title and subtitle. Finally, in a bar chart usually bars go on the vertical direction,
|
|
1650
|
-
but in this graph we want the bars to be horizontally
|
|
2258
|
+
but in this graph we want the bars to be horizontally laid so we add 'coord\_flip'.
|
|
1651
2259
|
|
|
1652
2260
|
```{ruby diverging_bar, fig.width = 9.1, fig.height = 6.5}
|
|
1653
2261
|
require 'ggplot'
|
|
@@ -1665,7 +2273,7 @@ puts mtcars.ggplot(E.aes(x: :car_name, y: :mpg_z, label: :mpg_z)) +
|
|
|
1665
2273
|
# Coding with Tidyverse
|
|
1666
2274
|
|
|
1667
2275
|
In R, and when coding with 'tidyverse', arguments to a function are usually not
|
|
1668
|
-
*
|
|
2276
|
+
*referentially transparent*. That is, you can’t replace a value with a seemingly equivalent
|
|
1669
2277
|
object that you’ve defined elsewhere. To see the problem, let's first define a data frame:
|
|
1670
2278
|
|
|
1671
2279
|
```{ruby df}
|
|
@@ -1681,17 +2289,17 @@ filter(df, my_var == 1)
|
|
|
1681
2289
|
```
|
|
1682
2290
|
It generates the following error: "object 'x' not found.
|
|
1683
2291
|
|
|
1684
|
-
However, in Galaaz, arguments are
|
|
1685
|
-
code
|
|
2292
|
+
However, in Galaaz, arguments are referentially transparent as can be seen by the
|
|
2293
|
+
code below. Note initially that 'my_var = R[:x]' will not give the error "object 'x' not found"
|
|
1686
2294
|
since ':x' is treated as an expression and assigned to my\_var. Then when doing (my\_var.eq 1),
|
|
1687
|
-
my\_var is a variable that resolves to ':x' and it becomes equivalent to (:x.eq 1) which is
|
|
2295
|
+
my\_var is a variable that resolves to ':x' and it becomes equivalent to (R[:x].eq 1) which is
|
|
1688
2296
|
what we want.
|
|
1689
2297
|
|
|
1690
2298
|
```{ruby my_var}
|
|
1691
|
-
my_var = :x
|
|
2299
|
+
my_var = R[:x]
|
|
1692
2300
|
puts df.filter(my_var.eq 1)
|
|
1693
2301
|
```
|
|
1694
|
-
As stated by
|
|
2302
|
+
As stated by Hadley
|
|
1695
2303
|
|
|
1696
2304
|
> dplyr code is ambiguous. Depending on what variables are defined where,
|
|
1697
2305
|
> filter(df, x == y) could be equivalent to any of:
|
|
@@ -1703,9 +2311,9 @@ df[x == df$y, ]
|
|
|
1703
2311
|
df[x == y, ]
|
|
1704
2312
|
```
|
|
1705
2313
|
In galaaz this ambiguity does not exist, filter(df, x.eq y) is not a valid expression as
|
|
1706
|
-
expressions are build with symbols. In doing filter(df, :x.eq y) we are looking for elements
|
|
2314
|
+
expressions are build with symbols. In doing filter(df, R[:x].eq y) we are looking for elements
|
|
1707
2315
|
of the 'x' column that are equal to a previously defined y variable. Finally in
|
|
1708
|
-
filter(df, :x.eq :y) we are looking for elements in which the 'x' column value is equal to
|
|
2316
|
+
filter(df, R[:x].eq R[:y]) we are looking for elements in which the 'x' column value is equal to
|
|
1709
2317
|
the 'y' column value. This can be seen in the following two chunks of code:
|
|
1710
2318
|
|
|
1711
2319
|
```{ruby disamb1}
|
|
@@ -1713,13 +2321,13 @@ y = 1
|
|
|
1713
2321
|
x = 2
|
|
1714
2322
|
|
|
1715
2323
|
# looking for values where the 'x' column is equal to the 'y' column
|
|
1716
|
-
puts df.filter(:x.eq :y)
|
|
2324
|
+
puts df.filter(R[:x].eq R[:y])
|
|
1717
2325
|
```
|
|
1718
2326
|
|
|
1719
2327
|
```{ruby disamb2}
|
|
1720
2328
|
# looking for values where the 'x' column is equal to the 'y' variable
|
|
1721
2329
|
# in this case, the number 1
|
|
1722
|
-
puts df.filter(:x.eq y)
|
|
2330
|
+
puts df.filter(R[:x].eq y)
|
|
1723
2331
|
```
|
|
1724
2332
|
## Writing a function that applies to different data sets
|
|
1725
2333
|
|
|
@@ -1746,11 +2354,12 @@ Unfortunately, in R, this function can fail silently if one of the variables isn
|
|
|
1746
2354
|
in the data frame, but is present in the global environment. We will not go through here how
|
|
1747
2355
|
to solve this problem in R.
|
|
1748
2356
|
|
|
1749
|
-
In Galaaz the method mutate_y
|
|
2357
|
+
In Galaaz the method mutate_y below will work fine and will never fail silently.
|
|
1750
2358
|
|
|
1751
2359
|
```{ruby mutate_y, warning=FALSE}
|
|
1752
2360
|
def mutate_y(df)
|
|
1753
|
-
|
|
2361
|
+
# Mutate column names are Ruby kwargs (y: …). Use .assign only for R `<-` expressions.
|
|
2362
|
+
df.mutate(y: R[:a] + R[:x])
|
|
1754
2363
|
end
|
|
1755
2364
|
```
|
|
1756
2365
|
Here we create a data frame that has only one column named 'x':
|
|
@@ -1760,8 +2369,8 @@ df1 = R.data__frame(x: (1..3))
|
|
|
1760
2369
|
puts df1
|
|
1761
2370
|
```
|
|
1762
2371
|
|
|
1763
|
-
Note that method mutate_y will fail
|
|
1764
|
-
in the scope of the method. Variable 'a' has no relationship with the symbol
|
|
2372
|
+
Note that method mutate_y will fail independently from the fact that variable 'a' is defined and
|
|
2373
|
+
in the scope of the method. Variable 'a' has no relationship with the symbol `R[:a]` used in the
|
|
1765
2374
|
definition of 'mutate\_y' above:
|
|
1766
2375
|
|
|
1767
2376
|
```{ruby call_mutate_y, warning = FALSE}
|
|
@@ -1770,12 +2379,13 @@ mutate_y(df1)
|
|
|
1770
2379
|
```
|
|
1771
2380
|
## Different expressions
|
|
1772
2381
|
|
|
1773
|
-
Let's move to the next problem as presented by
|
|
2382
|
+
Let's move to the next problem as presented by Hadley where trying to write a function in R
|
|
1774
2383
|
that will receive two argumens, the first a variable and the second an expression is not trivial.
|
|
1775
|
-
|
|
2384
|
+
Below we create a data frame and we want to write a function that groups data by a variable and
|
|
1776
2385
|
summarises it by an expression:
|
|
1777
2386
|
|
|
1778
2387
|
```{r diff_expr}
|
|
2388
|
+
library(dplyr)
|
|
1779
2389
|
set.seed(123)
|
|
1780
2390
|
|
|
1781
2391
|
df <- data.frame(
|
|
@@ -1800,7 +2410,7 @@ d2 <- df %>%
|
|
|
1800
2410
|
as.data.frame(d2)
|
|
1801
2411
|
```
|
|
1802
2412
|
|
|
1803
|
-
As shown by
|
|
2413
|
+
As shown by Hadley, one might expect this function to do the trick:
|
|
1804
2414
|
|
|
1805
2415
|
```{r diff_exp_fnc}
|
|
1806
2416
|
my_summarise <- function(df, group_var) {
|
|
@@ -1815,32 +2425,32 @@ my_summarise <- function(df, group_var) {
|
|
|
1815
2425
|
|
|
1816
2426
|
In order to solve this problem, coding with dplyr requires the introduction of many new concepts
|
|
1817
2427
|
and functions such as 'quo', 'quos', 'enquo', 'enquos', '!!' (bang bang), '!!!' (triple bang).
|
|
1818
|
-
Again, we'll leave to
|
|
2428
|
+
Again, we'll leave to Hadley the explanation on how to use all those functions.
|
|
1819
2429
|
|
|
1820
2430
|
Now, let's try to implement the same function in galaaz. The next code block first prints the
|
|
1821
|
-
'df' data frame defined previously in R (to access an R variable from Galaaz, we use the
|
|
1822
|
-
operator
|
|
2431
|
+
'df' data frame defined previously in R (to access an R variable from Galaaz, we use the tilde
|
|
2432
|
+
operator `~` applied to the R variable name as a symbol, e.g. `:df`).
|
|
1823
2433
|
|
|
1824
2434
|
```{ruby r_dataframe}
|
|
1825
|
-
puts
|
|
2435
|
+
puts ~R[:df]
|
|
1826
2436
|
```
|
|
1827
2437
|
|
|
1828
2438
|
We then create the 'my_summarize' method and call it passing the R data frame and
|
|
1829
|
-
the group by variable ':g1':
|
|
2439
|
+
the group by variable 'R[:g1]':
|
|
1830
2440
|
|
|
1831
2441
|
```{ruby diff_exp_ruby_func}
|
|
1832
2442
|
def my_summarize(df, group_var)
|
|
1833
2443
|
df.group_by(group_var).
|
|
1834
|
-
summarize(a: :a.mean)
|
|
2444
|
+
summarize(a: R[:a].mean)
|
|
1835
2445
|
end
|
|
1836
2446
|
|
|
1837
|
-
puts my_summarize(:df, :g1)
|
|
2447
|
+
puts my_summarize(~R[:df], R[:g1])
|
|
1838
2448
|
```
|
|
1839
2449
|
|
|
1840
2450
|
It works!!! Well, let's make sure this was not just some coincidence
|
|
1841
2451
|
|
|
1842
2452
|
```{ruby group_g2}
|
|
1843
|
-
puts my_summarize(:df, :g2)
|
|
2453
|
+
puts my_summarize(~R[:df], R[:g2])
|
|
1844
2454
|
```
|
|
1845
2455
|
|
|
1846
2456
|
Great, everything is fine! No magic, no new functions, no complexities, just normal, standard Ruby
|
|
@@ -1852,7 +2462,7 @@ In the previous section we've managed to get rid of all NSE formulation for a si
|
|
|
1852
2462
|
does this remain true for more complex examples, or will the Galaaz way prove inpractical for
|
|
1853
2463
|
more complex code?
|
|
1854
2464
|
|
|
1855
|
-
In the next example
|
|
2465
|
+
In the next example Hadley proposes us to write a function that given an expression such as 'a'
|
|
1856
2466
|
or 'a * b', calculates three summaries. What we want a function that does the same as these R
|
|
1857
2467
|
statements:
|
|
1858
2468
|
|
|
@@ -1881,9 +2491,9 @@ def my_summarise2(df, expr)
|
|
|
1881
2491
|
)
|
|
1882
2492
|
end
|
|
1883
2493
|
|
|
1884
|
-
puts my_summarise2((
|
|
2494
|
+
puts my_summarise2((~R[:df]), :a)
|
|
1885
2495
|
puts "\n"
|
|
1886
|
-
puts my_summarise2((
|
|
2496
|
+
puts my_summarise2((~R[:df]), R[:a] * R[:b])
|
|
1887
2497
|
```
|
|
1888
2498
|
|
|
1889
2499
|
Once again, there is no need to use any special theory or functions. The only point to be
|
|
@@ -1891,7 +2501,7 @@ careful about is the use of 'E' to build expressions from functions 'mean', 'sum
|
|
|
1891
2501
|
|
|
1892
2502
|
## Different input and output variable
|
|
1893
2503
|
|
|
1894
|
-
Now the next challenge presented by
|
|
2504
|
+
Now the next challenge presented by Hadley is to vary the name of the output variables based on
|
|
1895
2505
|
the received expression. So, if the input expression is 'a', we want our data frame columns to
|
|
1896
2506
|
be named 'mean\_a' and 'sum\_a'. Now, if the input expression is 'b', columns
|
|
1897
2507
|
should be named 'mean\_b' and 'sum\_b'.
|
|
@@ -1917,7 +2527,7 @@ mutate(df, mean_b = mean(b), sum_b = sum(b))
|
|
|
1917
2527
|
#> 4 2 2 5 4 3 15
|
|
1918
2528
|
#> # … with 1 more row
|
|
1919
2529
|
```
|
|
1920
|
-
In order to solve this problem in R,
|
|
2530
|
+
In order to solve this problem in R, Hadley needs to introduce some more new functions and notations:
|
|
1921
2531
|
'quo_name' and the ':=' operator from package 'rlang'
|
|
1922
2532
|
|
|
1923
2533
|
Here is our Ruby code:
|
|
@@ -1931,9 +2541,9 @@ def my_mutate(df, expr)
|
|
|
1931
2541
|
sum_name => E.sum(expr))
|
|
1932
2542
|
end
|
|
1933
2543
|
|
|
1934
|
-
puts my_mutate((
|
|
2544
|
+
puts my_mutate((~R[:df]), :a)
|
|
1935
2545
|
puts "\n"
|
|
1936
|
-
puts my_mutate((
|
|
2546
|
+
puts my_mutate((~R[:df]), :b)
|
|
1937
2547
|
```
|
|
1938
2548
|
It really seems that "Non Standard Evaluation" is actually quite standard in Galaaz! But, you
|
|
1939
2549
|
might have noticed a small change in the way the arguments to the mutate method were called.
|
|
@@ -1945,7 +2555,7 @@ and variable mean\_name is not followed by ':' but by '=>'. This is standard Ru
|
|
|
1945
2555
|
|
|
1946
2556
|
## Capturing multiple variables
|
|
1947
2557
|
|
|
1948
|
-
Moving on with new complexities,
|
|
2558
|
+
Moving on with new complexities, Hadley proposes us to solve the problem in which the
|
|
1949
2559
|
summarise function will receive any number of grouping variables.
|
|
1950
2560
|
|
|
1951
2561
|
This again is quite standard Ruby. In order to receive an undefined number of paramenters
|
|
@@ -1957,7 +2567,7 @@ def my_summarise3(df, *group_vars)
|
|
|
1957
2567
|
summarise(a: E.mean(:a))
|
|
1958
2568
|
end
|
|
1959
2569
|
|
|
1960
|
-
puts my_summarise3((
|
|
2570
|
+
puts my_summarise3((~R[:df]), R[:g1], R[:g2])
|
|
1961
2571
|
```
|
|
1962
2572
|
|
|
1963
2573
|
## Why does R require NSE and Galaaz does not?
|
|
@@ -1975,7 +2585,7 @@ In Ruby, there is no lazy evaluation of parameters and 'a' is always a variable
|
|
|
1975
2585
|
Variables assume their value as soon as they are used, so 'x = a' is immediately evaluate and
|
|
1976
2586
|
variable 'x' will receive the value of variable 'a' as soon as the Ruby statement is executed.
|
|
1977
2587
|
Ruby also provides the notion of a symbol; ':a' is a symbol and does not evaluate to anything.
|
|
1978
|
-
Galaaz uses Ruby symbols to build expressions that are not bound to anything: ':a.eq :b' is
|
|
2588
|
+
Galaaz uses Ruby symbols to build expressions that are not bound to anything: 'R[:a].eq R[:b]' is
|
|
1979
2589
|
clearly an expression and has no relationship whatsoever with the statment 'a = b'. By using
|
|
1980
2590
|
symbols, variables and expressions all the possible ambiguities that are found in R are
|
|
1981
2591
|
eliminated in Galaaz.
|
|
@@ -1985,7 +2595,7 @@ of input they are expecting, they might be expecting regular variables or they m
|
|
|
1985
2595
|
expecting expressions and the R function will know how to deal with an input of the form
|
|
1986
2596
|
'a = b', now for the Ruby developer it might not be immediately clear if it should call the
|
|
1987
2597
|
function passing the value 'true' if variable 'a' is equal to variable 'b' or if it should
|
|
1988
|
-
call the function passing the expression ':a.eq :b'.
|
|
2598
|
+
call the function passing the expression 'R[:a].eq R[:b]'.
|
|
1989
2599
|
|
|
1990
2600
|
|
|
1991
2601
|
## Advanced dplyr features
|
|
@@ -2008,12 +2618,13 @@ In the following examples, we show the use of functions 'group\_by\_at', 'summar
|
|
|
2008
2618
|
features of characters in the Starwars movies:
|
|
2009
2619
|
|
|
2010
2620
|
```{ruby starwars}
|
|
2011
|
-
puts (
|
|
2621
|
+
puts (~R[:starwars]).head
|
|
2012
2622
|
```
|
|
2013
|
-
The grouped_mean function
|
|
2623
|
+
The grouped_mean function below will receive a grouping variable and calculate summaries for
|
|
2014
2624
|
the value\_variables given:
|
|
2015
2625
|
|
|
2016
2626
|
```{r grouped_mean}
|
|
2627
|
+
library(dplyr)
|
|
2017
2628
|
grouped_mean <- function(data, grouping_variables, value_variables) {
|
|
2018
2629
|
data %>%
|
|
2019
2630
|
group_by_at(grouping_variables) %>%
|
|
@@ -2035,24 +2646,26 @@ def grouped_mean(data, grouping_variables, value_variables)
|
|
|
2035
2646
|
data.
|
|
2036
2647
|
group_by_at(grouping_variables).
|
|
2037
2648
|
mutate(count: E.n).
|
|
2038
|
-
summarise_at(E.c(value_variables, "count"),
|
|
2649
|
+
summarise_at(E.c(value_variables, "count"), ~R[:mean], na__rm: true).
|
|
2039
2650
|
rename_at(value_variables, E.funs(E.paste0("mean_", value_variables)))
|
|
2040
2651
|
end
|
|
2041
2652
|
|
|
2042
|
-
puts grouped_mean((
|
|
2653
|
+
puts grouped_mean((~R[:starwars]), "eye_color", E.c("mass", "birth_year"))
|
|
2043
2654
|
```
|
|
2044
2655
|
|
|
2045
|
-
|
|
2046
|
-
|
|
2047
|
-
|
|
2656
|
+
The examples above cover programmatic dplyr with string column names and `_at` helpers. The same
|
|
2657
|
+
Galaaz patterns (symbols, `E.*` for expression-safe functions, and Ruby methods on R-backed objects)
|
|
2658
|
+
extend to other tidyverse workflows; consult R package documentation for function-specific
|
|
2659
|
+
arguments.
|
|
2048
2660
|
|
|
2049
2661
|
# Contributing
|
|
2050
2662
|
|
|
2051
2663
|
* Fork it
|
|
2052
|
-
* Create your feature branch (git checkout -b my-new-feature)
|
|
2053
|
-
* Write
|
|
2054
|
-
|
|
2055
|
-
*
|
|
2056
|
-
*
|
|
2664
|
+
* Create your feature branch (`git checkout -b my-new-feature`)
|
|
2665
|
+
* Write tests — use **`bin/run_rspec`** or **`bin/run_all_rspec`** with **JRuby** so JVM flags and
|
|
2666
|
+
the load path match **`docs/testing.md`**
|
|
2667
|
+
* Commit your changes (`git commit -am 'Add some feature'`)
|
|
2668
|
+
* Push to the branch (`git push origin my-new-feature`)
|
|
2669
|
+
* Open a pull request
|
|
2057
2670
|
|
|
2058
2671
|
# References
|