rsyntaxtree 1.13.2 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,7 +13,7 @@ require_relative "color_names"
13
13
 
14
14
  module RSyntaxTree
15
15
  class Element
16
- attr_accessor :rule_name, :id, :parent, :type, :level, :width, :height, :content, :content_width, :text_width, :content_height, :horizontal_indent, :vertical_indent, :triangle, :enclosure, :children, :font, :fontsize, :contains_phrase, :path, :color, :raw_content, :region, :region_color
16
+ attr_accessor :ink_top, :ink_bottom, :rule_name, :label_with_rule_name, :id, :parent, :type, :level, :width, :height, :content, :content_width, :text_width, :content_height, :horizontal_indent, :vertical_indent, :triangle, :enclosure, :children, :font, :fontsize, :contains_phrase, :path, :color, :raw_content, :region, :region_color
17
17
 
18
18
  # names_a_rule says the content is a mother's label rather than a leaf's
19
19
  # text. Only a mother has a step under it for a name to sit beside, and a
@@ -42,7 +42,11 @@ module RSyntaxTree
42
42
  @fontset = fontset
43
43
  @fontsize = fontsize
44
44
  # In a derivation the label may carry the name of the rule that produced
45
- # it, written after a column break: `S/NP\t\>B`. The name belongs beside
45
+ # it, written after a column break: `S/NP\t>B`. The name is taken out
46
+ # before the label is read as markup, so it is drawn exactly as written —
47
+ # `>` and `<` need no escaping there, and escaping them puts the backslash
48
+ # in the figure. (This example carried one, which is the opposite of what
49
+ # the gallery's derivations do.) The name belongs beside
46
50
  # the rule rather than beside the result, so it comes out of the label
47
51
  # here, before the label is measured, and BaseGraph draws it at the end
48
52
  # of the rule.
@@ -51,6 +55,11 @@ module RSyntaxTree
51
55
  name = at && content[(at + 2)..].to_s.strip
52
56
  if name && !name.empty?
53
57
  @rule_name = name
58
+ # Kept whole, because whether this node has daughters — and so
59
+ # whether it names a rule at all — is not known until the tree
60
+ # is built. StringParser puts the label back if it turns out
61
+ # there was no step for the name to belong to.
62
+ @label_with_rule_name = content
54
63
  content = content[0...at]
55
64
  end
56
65
  end
@@ -300,6 +309,10 @@ module RSyntaxTree
300
309
  row_holds_text = content[:elements].any? do |e|
301
310
  (e[:decoration] & [:box, :circle, :bar]).empty? && !e[:text].to_s.strip.empty?
302
311
  end
312
+ # True until something is drawn in the current column. A tabstop
313
+ # opens the next one, so a block that starts a column is at the
314
+ # start of a cell however many columns came before it.
315
+ cell_start = true
303
316
  content[:elements].each do |e|
304
317
  # A nested matrix is measured by the same code one level down, and
305
318
  # reports the size of the block it will occupy in this row: its own
@@ -308,7 +321,18 @@ module RSyntaxTree
308
321
  inner = measure_lines(e[:matrix], nested: true)
309
322
  e[:matrix_width] = inner[:width]
310
323
  e[:matrix_height] = inner[:height]
311
- e[:width] = inner[:width] + matrix_bracket_room * 2
324
+ # A block that opens a cell needs nothing in front of it: the
325
+ # column it starts is already held clear. One that follows
326
+ # something in the same cell — the tag on a shared value, as in
327
+ # AGR |1| [ ... ] — was drawn with its bracket on the ink before
328
+ # it, because a run is spaced for the glyph that comes next and
329
+ # a bracket is not a glyph.
330
+ # As much room in front of the bracket as the bracket keeps
331
+ # inside it, so the two sides of it look alike. Half as much
332
+ # left it looking crowded from the outside.
333
+ e[:matrix_lead] = cell_start ? 0 : matrix_bracket_room
334
+ cell_start = false
335
+ e[:width] = inner[:width] + matrix_bracket_room * 2 + e[:matrix_lead]
312
336
  # Two separate allowances. The block is padded inside its own
313
337
  # brackets, above and below, and that padding is part of the row.
314
338
  # The gap that keeps the block clear of the rows either side is
@@ -326,6 +350,14 @@ module RSyntaxTree
326
350
  next
327
351
  end
328
352
 
353
+ # A tabstop opens the next column; anything else puts ink in the
354
+ # one being filled.
355
+ if e[:decoration].to_a.include?(:tabstop)
356
+ cell_start = true
357
+ elsif !e[:text].to_s.empty?
358
+ cell_start = false
359
+ end
360
+
329
361
  text = e[:text]
330
362
  # Handle escaped square brackets
331
363
  text = text.gsub('\\[', '[')
@@ -291,12 +291,18 @@ module RSyntaxTree
291
291
  connector_height: @params[:vheight],
292
292
  horizontal_spacing: @params[:hspacing] || 1.0,
293
293
  line_width: @params[:linewidth],
294
- # The layout scale the figure was drawn on. symmetrize is kept
295
- # for readers of older files; tidy: "symmetric" supersedes it.
294
+ # The layout scale the figure was drawn on. Files written
295
+ # before 2.0 carry a separate symmetrize flag beside it; the
296
+ # scale is what it meant, and it is all that is written now.
296
297
  tidy: @params[:tidy] || "off",
297
- symmetrize: @params[:symmetrize],
298
298
  mirror: @params[:mirror] == true,
299
299
  direction: @params[:direction] || "ttb",
300
+ # Degrees, positive leaning the top right. The geometry below
301
+ # is the unsheared layout — the shear is applied at render — so
302
+ # a reader who wants the drawn look applies it, and one who
303
+ # wants the layout ignores it.
304
+ shear: (@params[:shear] || 0).to_f,
305
+ vmargin: (@params[:vmargin] || 0.4).to_f,
300
306
  # Which of the two readings of a hyphen the input was parsed
301
307
  # under. The input is recorded verbatim above, and the same
302
308
  # string means different things under the two, so a reader
@@ -5,8 +5,13 @@ require 'parslet'
5
5
  class MarkupParser < Parslet::Parser
6
6
  rule(:cr) { str('\\n') }
7
7
  rule(:eof) { any.absent? }
8
- rule(:border) { match('[^\-]').absent? >> str('-').repeat(3).as(:border) >> (eof | cr) }
9
- rule(:bborder) { match('[^=]').absent? >> str('=').repeat(3).as(:bborder) >> (eof | cr) }
8
+ rule(:border) { rule_line >> (eof | cr) }
9
+ rule(:bborder) { double_rule_line >> (eof | cr) }
10
+ # The rule itself, without what ends the line. A matrix ends its rows on
11
+ # its own closing delimiter as well as on a line break, so it supplies its
12
+ # own terminator; folding one in here would have the two consume it twice.
13
+ rule(:rule_line) { match('[^\-]').absent? >> str('-').repeat(3).as(:border) }
14
+ rule(:double_rule_line) { match('[^=]').absent? >> str('=').repeat(3).as(:bborder) }
10
15
 
11
16
  rule(:brectangle) { str('###') }
12
17
  rule(:rectangle) { str('##') }
@@ -15,7 +20,14 @@ class MarkupParser < Parslet::Parser
15
20
 
16
21
  # Color specification: @colorname: or @#hexcode:
17
22
  rule(:color_name) { match('[a-zA-Z]').repeat(1) }
18
- rule(:color_hex) { str('#') >> match('[0-9a-fA-F]').repeat(3, 6) }
23
+ # Three digits or six, which is what every message about colour in this
24
+ # codebase already says. The rule used to be written as "three to six", so
25
+ # four and five passed the parser, passed the validator (which does not look
26
+ # at a value beginning with '#' at all), and went into the SVG — where
27
+ # librsvg cannot read them and draws the label black, with nothing reported.
28
+ # Six first: on a six-digit value the three-digit branch matches and then
29
+ # the ':' is not there to be found.
30
+ rule(:color_hex) { str('#') >> (match('[0-9a-fA-F]').repeat(6, 6) | match('[0-9a-fA-F]').repeat(3, 3)) }
19
31
  rule(:color_spec) { str('@') >> (color_hex | color_name).as(:color_value) >> str(':') }
20
32
 
21
33
  # Region shade: '%' marks the node so that the whole subtree it governs
@@ -72,7 +84,7 @@ class MarkupParser < Parslet::Parser
72
84
  # bare brackets would be read as tree structure and bare parentheses appear
73
85
  # in labels too often to claim.
74
86
  rule(:matrix) { str('#(') >> matrix_line.repeat(1).as(:matrix) >> str('#)') }
75
- rule(:matrix_line) { matrix_markup.repeat(1).as(:line) >> (cr | str('#)').present?) }
87
+ rule(:matrix_line) { (rule_line | double_rule_line | matrix_markup.repeat(1).as(:line)) >> (cr | str('#)').present?) }
76
88
  rule(:matrix_markup) { (matrix | tabstop | matrix_text | decoration | shape | bstroke) }
77
89
  # Text inside a matrix stops at the closing delimiter as well.
78
90
  rule(:matrix_text) { (escaped | (str('#)').absent? >> non_escaped)).repeat(1).as(:text) }
@@ -1,52 +1,118 @@
1
- RSyntaxTree notation, in brief.
1
+ RSyntaxTree notation, on one page.
2
2
 
3
3
  A tree is labeled brackets: [S [NP the cat] [VP sat]]. The label follows the
4
- opening bracket; children follow the label.
5
-
6
- Within a label:
7
- - `\n` breaks a line; `\t` separates columns, and every line is cut at the
8
- same points, so the columns line up down the label. (`\t` is the two
9
- characters backslash and t, not a real tab.)
10
- - `#label` draws square brackets around the whole label; `##label` a
11
- rectangle. `#( ... #)` nests a matrix inside a label, to any depth.
12
- - `*italic*`, `**bold**`, `x_sub_`, `x__super__`, `|x|` in a box, `{x}` in a
13
- circle, `---` a horizontal rule.
14
- - `+n` on two nodes links them with a movement arrow; `+>n` puts the
15
- arrowhead on that end.
16
-
17
- Four traps, because these characters already mean something:
18
- - `<` and `>` are never angle brackets in this notation: `<>` is one space,
19
- `<3>` is three. Wherever linguistics uses angle brackets — a list ⟨NP⟩,
20
- an argument structure like 'hand⟨SUBJ,OBJ⟩', anything — write the
21
- characters ⟨ and ⟩ themselves (U+27E8 and U+27E9):
22
- SPR\t⟨<>NP<>⟩ PRED\t'hand⟨SUBJ,OBJ⟩'
23
- - `-` opens and closes an underline, so a bare hyphen in ANY word is an
24
- error: V-bar, f-structure, HEAD-DTR. Write `V'`, or escape the hyphen:
25
- `V\-bar`, `f\-structure`, `HEAD\-DTR`.
26
- - A raw space is safe in a one-line label, and unreliable in a label that
27
- has `\n` or `\t` in it: there it splits a value that carries markup, and
28
- it breaks a matrix nested with `#( ... #)`. Since matrices are exactly the
29
- multi-line case, write a space as `<>` inside any label with columns:
30
- 'a<>toy', not 'a toy'.
31
- - Parentheses are not brackets, and this is silent: `(S (NP ...))` raises no
32
- error and draws one leaf with that text in it. Convert it first.
33
-
34
- An attribute-value matrix is columns plus an enclosure:
35
-
36
- [#*word*\
37
- HEAD\t*verb*\
38
- SPR\t⟨<>NP<>⟩\
39
- COMPS\t⟨<>⟩
40
- ]
4
+ opening bracket; children follow the label. Parentheses are read as Penn
5
+ Treebank notation and converted: (S (NP the cat) (VP sat)) draws the same tree.
6
+
7
+ ## What already means something
8
+
9
+ The characters that bite, first. A mistake the tool can see is refused with a
10
+ hint; everything below is taken silently, and the figure that draws is not the
11
+ figure that was meant.
12
+
13
+ - `<>` is one space and `<3>` is three, everywhere — never angle brackets.
14
+ Where linguistics wants the brackets themselves, write ⟨ and ⟩ (U+27E8/9):
15
+ SPR\t⟨<>NP<>⟩, 'hand⟨SUBJ,OBJ⟩'. A label written <NP> is refused with this
16
+ advice; a stray <3> simply draws three spaces. No text face carries these
17
+ two characters, so whatever the machine falls back to draws them — they
18
+ will not quite match the letters beside them, and how far off they look
19
+ depends on the fonts installed.
20
+ - A pair of hyphens underlines what stands between them: well-made-word draws
21
+ "made" underlined. (A lone hyphen is refused with a hint.) Write `\-` for a
22
+ hyphen — V\-bar, HEAD\-DTR — or set the hyphen option to literal.
23
+ - A raw space is safe in a one-line label and unreliable in a label that has
24
+ `\n` or `\t` in it: there it can split a value that carries markup, and it
25
+ breaks a nested #( ... #) matrix. Inside any label with columns, write a
26
+ space as `<>`: 'a<>toy', not 'a toy'.
27
+ - A straight apostrophe is typeset as the curly ’ (U+2019).
28
+ - A backslash takes the character after it, whatever it is: \q draws q, and
29
+ C:\path draws C:path. A backslash itself is \\.
30
+ - The prefixes of a label compose in one order, and only this one:
31
+ `^` → `#`/`##`/`###` → `%` → `@color:`. So ^#%@red:NP is read whole, while
32
+ @red:%NP leaves the % as a literal character and silently drops the shade.
33
+ - A rule name (for derivations) is written after the label's last `\t`, as in
34
+ [S/NP\t>B ...], and needs the derivation option ON — with it off the name is
35
+ drawn as a column. The name is taken out before markup runs, so `>` and `<`
36
+ need no escaping there and `\>` would put the backslash in the figure. A
37
+ label that also carries `\n` keeps all its columns instead.
38
+
39
+ ## What there is
40
+
41
+ Structure:
42
+
43
+ | what | how it is written |
44
+ |----------------------------------------|-------------------|
45
+ | a node with children | [S [NP the cat] [VP sat]] |
46
+ | a phrase under one leaf (triangle) | [NP a toy] — one word, forced: [NP ^cats] |
47
+ | an invisible joint, to align leaves | a label of only <>, as in [X [<> [Y y]] [Z z]] |
48
+ | a movement path (dashed) | [S [NP a+1] [VP [V b] [NP c+1]]] |
49
+ | the same, with an arrowhead | write +>1 on the end the arrow points at |
50
+ | an extra straight connector | +-1 on two nodes; +->1 for an arrowhead |
41
51
 
42
- A value can be another matrix. Nest it with `#( ... #)`, never with square
43
- brackets — those are read as tree structure:
52
+ Inside a label:
53
+
54
+ | what | how it is written |
55
+ |----------------------------------------|-------------------|
56
+ | italic, bold, both | *x*, **x**, ***x*** |
57
+ | subscript, superscript | x_i_, x__2__ |
58
+ | small capitals | H___EAD___ |
59
+ | overline, underline, strikethrough | =x=, -x-, ~x~ |
60
+ | a line break; a blank line | a\nb (or a\ b); \n\n |
61
+ | columns, aligned down the label | HEAD\t*verb*\nSPR\t⟨<>⟩ |
62
+ | a horizontal rule across the label | a line of --- (or === for a double rule) |
63
+ | boxed, circled text | |1|, {2} — more than one character draws a capsule |
64
+ | empty and hatched boxes and circles | ||, {}, |/|, {/} |
65
+ | bar and arrows as symbols | --, ->, <-, <-> — bold with *...*, as in *->* |
66
+
67
+ Around a label:
68
+
69
+ | what | how it is written |
70
+ |----------------------------------------|-------------------|
71
+ | square brackets, rectangle, bold | #NP, ##NP, ###NP |
72
+ | a feature matrix as a value | #(HEAD\tnoun#), nested to any depth |
73
+ | a whole label that is one matrix | [#(CAT\tS#) [#(CAT\tNP#) Kim] [#(CAT\tVP#) sleeps]] |
74
+ | a region shade behind the subtree | %NP, coloured: %@blue:NP |
75
+ | a colour | @red:NP, @#3af:NP, @#33aaff:NP (3 or 6 hex digits) |
76
+
77
+ An attribute-value matrix is columns plus an enclosure, and a value can be
78
+ another matrix — nested with #( ... #), never with square brackets, which are
79
+ read as tree structure:
44
80
 
45
81
  [#PRED\t'hand⟨SUBJ,OBJ⟩'\
46
82
  TENSE\tpast\
47
83
  SUBJ\t#(PRED\t'David'#)
48
84
  ]
49
85
 
50
- A movement arrow links the two nodes that carry the same number:
86
+ Options (the command line spells them --like-this):
87
+
88
+ | option | what it decides |
89
+ |------------|-----------------|
90
+ | format | png, svg, pdf, tikz, lsif |
91
+ | fontstyle | sans, serif, mono, cjk |
92
+ | fontsize | 6–26 |
93
+ | color | modern, traditional, off, gray |
94
+ | linewidth | every line's weight, 0.5–3.0, as a ratio of the font size |
95
+ | leafstyle | auto, bar, nothing — what joins a node to its leaf |
96
+ | direction | ttb, ltr, btt |
97
+ | mirror | flip the finished layout, for RTL scripts |
98
+ | tidy | off, symmetric, low, medium, high — how tightly subtrees pack |
99
+ | hspacing | how far apart sisters sit, 0.5–3.0 |
100
+ | vheight | how far apart levels sit, 0.5–5.0 |
101
+ | polyline | draw each connector as two right angles rather than a slanted line |
102
+ | hide_default_connectors | draw no parent-to-child lines, leaving only the ones written with +-n — how a lattice or a network is drawn |
103
+ | transparent | leave the background clear instead of painting it white |
104
+ | derivation | one rule across the daughters, categorial-grammar style |
105
+ | hyphen | markup (hyphens underline) or literal (hyphens are hyphens) |
106
+ | shear | tilt the drawn figure: degrees, −45 to 45, positive leans the top right |
107
+ | shear_plane | the plane behind a sheared figure: on (grey), off, or a colour; never drawn on a transparent background |
108
+ | vmargin | clearance between a label and its connectors, 0.1–1.0, the same above and below |
109
+
110
+ hspacing and vheight are named for the top-to-bottom layout, and follow the
111
+ tree rather than the page: with direction ltr, sisters stack downwards and
112
+ hspacing is what separates them, while vheight runs across.
113
+
114
+ The full manual explains each of these with figures, and every example in the
115
+ gallery is written out beside the figure it draws:
51
116
 
52
- [S [NP what+>1] [VP [V see] [NP+1 t]]]
117
+ https://yohasebe.github.io/rsyntaxtree/documentation
118
+ https://yohasebe.github.io/rsyntaxtree/examples
@@ -1,4 +1,4 @@
1
- RSyntaxTree examples: 76 trees, every one verified to draw.
1
+ RSyntaxTree examples: 80 trees, every one verified to draw.
2
2
 
3
3
  Each is the input behind a figure in the gallery at https://yohasebe.github.io/rsyntaxtree/examples.
4
4
  The settings line names the options the gallery records for that figure;
@@ -8,10 +8,10 @@ here through the same parser that draws it, so each one is accepted.
8
8
  Notation reference: rsyntaxtree --notation
9
9
  Check an input without drawing: rsyntaxtree --validate "[S [NP a] [VP b]]"
10
10
 
11
- What the option names on a settings line mean is not in this file, nor in
12
- the reference, which leaves the options out on purpose. For those, see
13
- `rsyntaxtree --help` or the manual: https://yohasebe.github.io/rsyntaxtree/documentation. Everything at
14
- once, this file and the manual together: https://yohasebe.github.io/rsyntaxtree/llms-full.txt
11
+ What the option names on a settings line mean: one line each at the end
12
+ of the reference (`rsyntaxtree --notation`), the full story in the manual
13
+ at https://yohasebe.github.io/rsyntaxtree/documentation. Everything at once, this file and the manual
14
+ together: https://yohasebe.github.io/rsyntaxtree/llms-full.txt
15
15
 
16
16
 
17
17
  ## 000 — RSyntaxTree basic example
@@ -2521,3 +2521,95 @@ Source: Steedman 2000
2521
2521
  [(S\\NP)/NP hates]]]
2522
2522
  [NP London]]
2523
2523
  ```
2524
+
2525
+ ## 082 — Indo-European language family
2526
+
2527
+ Category: Historical Linguistics
2528
+ Settings: direction=ltr fontstyle=noto-serif hspacing=0.5 polyline=on vheight=5.0
2529
+
2530
+ ```
2531
+ [Proto\-Indo\-European
2532
+ [%@#4a7c59:Italic
2533
+ [Latin
2534
+ [Western [*Spanish*] [*Portuguese*] [*French*]]
2535
+ [Eastern [*Italian*] [*Romanian*]]]]
2536
+ [%@#3d6b9e:Germanic
2537
+ [West [*English*] [*German*] [*Dutch*]]
2538
+ [North [*Swedish*] [*Icelandic*]]]
2539
+ [%@#b5651d:Balto\-Slavic
2540
+ [Slavic [*Russian*] [*Polish*] [*Czech*]]
2541
+ [Baltic [*Lithuanian*] [*Latvian*]]]
2542
+ [%@#8e4585:Indo\-Iranian
2543
+ [Indo\-Aryan [*Hindi*] [*Bengali*]]
2544
+ [Iranian [*Persian*] [*Pashto*]]]
2545
+ [%@#7a6a9b:Hellenic [Ancient<>Greek [*Greek*]]]
2546
+ [%@#2f7d7d:Celtic [*Irish*] [*Welsh*]]]
2547
+ ```
2548
+
2549
+ ## 083 — Type-driven semantic composition
2550
+
2551
+ Category: Formal Semantics
2552
+ Settings: color=none fontstyle=noto-serif leafstyle=nothing
2553
+
2554
+ ```
2555
+ [S\n*t*\n**love**(**m**)(**j**)
2556
+ [DP\n*e*\n**j** [*John*]]
2557
+ [VP\n⟨*e*,*t*⟩\nλ*x*.**love**(**m**)(*x*)
2558
+ [V\n⟨*e*,⟨*e*,*t*⟩⟩\nλ*y*λ*x*.**love**(*y*)(*x*) [*loves*]]
2559
+ [DP\n*e*\n**m** [*Mary*]]]]
2560
+ ```
2561
+
2562
+ ## 084 — Typed feature structures with structure sharing
2563
+
2564
+ Category: Formal Grammar
2565
+ Settings: color=none fontstyle=noto-serif leafstyle=nothing
2566
+
2567
+ ```
2568
+ [#(*phrase*\
2569
+ ---\
2570
+ SYN\t#(HEAD\t|1|\
2571
+ VAL\t#(SPR\t⟨<>⟩#)#)#)
2572
+ [#(*phrase*\
2573
+ ---\
2574
+ SYN\t#(HEAD\t#(*noun*\
2575
+ ---\
2576
+ AGR\t|2|#(*3sing*\
2577
+ ---\
2578
+ PER\t3\
2579
+ NUM\t*sg*#)#)\
2580
+ VAL\t#(SPR\t⟨<>⟩#)#)#) [the<>dog]]
2581
+ [#(*word*\
2582
+ ---\
2583
+ SYN\t#(HEAD\t|1|#(*verb*\
2584
+ ---\
2585
+ AGR\t|2|#)\
2586
+ VAL\t#(SPR\t⟨<>*NP*<>⟩#)#)#) [barks]]]
2587
+ ```
2588
+
2589
+ ## 085 — A tree on a tilted plane
2590
+
2591
+ Category: General
2592
+ Settings: fontstyle=noto-serif shear=20 vheight=1.0
2593
+
2594
+ ```
2595
+ [S
2596
+ [NP
2597
+ [D the]
2598
+ [N man]
2599
+ ]
2600
+ [VP
2601
+ [V put]
2602
+ [NP
2603
+ [D the]
2604
+ [N book]
2605
+ ]
2606
+ [PP
2607
+ [P on]
2608
+ [NP
2609
+ [D the]
2610
+ [N table]
2611
+ ]
2612
+ ]
2613
+ ]
2614
+ ]
2615
+ ```
@@ -96,9 +96,44 @@ module RSyntaxTree
96
96
  end
97
97
  end
98
98
 
99
+ # Whether the raw space that split this token is what stopped it parsing.
100
+ # Asked of the parser rather than reasoned about: the whole token, with its
101
+ # spaces written as `<>`, either reads as one label or it does not.
102
+ def space_is_the_cause?(token, parent)
103
+ Element.new(-1, parent, token.gsub(" ", WHITESPACE_BLOCK), @level,
104
+ @fontset, @fontsize, @global, true)
105
+ true
106
+ rescue StandardError
107
+ false
108
+ end
109
+
99
110
  def parse
100
111
  make_tree(0);
101
112
  @elist.set_hierarchy
113
+ restore_rule_names_without_a_rule
114
+ end
115
+
116
+ # A rule name names the step that produced a node from its daughters. A node
117
+ # with no daughters is the product of no step, so what looked like a name is
118
+ # a column of the label like any other, and it goes back.
119
+ #
120
+ # Whether a node will have daughters is not known where the label is read —
121
+ # they arrive as later tokens — so the label is read first and put right
122
+ # here, once the tree is built. Left alone, turning the option on deleted a
123
+ # column: `[A\tfoo]` drew "A foo" with derivation off and "A" with it on,
124
+ # and said nothing about the difference.
125
+ def restore_rule_names_without_a_rule
126
+ @elist.elements.each_with_index do |e, i|
127
+ next if e.rule_name.nil? || e.rule_name.empty?
128
+ next unless e.children.empty?
129
+ next if e.label_with_rule_name.nil?
130
+
131
+ restored = Element.new(e.id, e.parent, e.label_with_rule_name,
132
+ e.level, @fontset, @fontsize, @global)
133
+ restored.children = e.children
134
+ restored.type = e.type
135
+ @elist.elements[i] = restored
136
+ end
102
137
  end
103
138
 
104
139
  def get_elementlist
@@ -215,7 +250,21 @@ module RSyntaxTree
215
250
  # parse, the likeliest story is that the space belongs
216
251
  # inside the label and cut a construct in two — say so,
217
252
  # unless a more specific cause is already known.
218
- raise e if e.code == :bare_hyphen
253
+ #
254
+ # Which story is right is not a thing to guess at: ask
255
+ # whether the space is the one that breaks it. Put the whole
256
+ # token back together with the spaces written as the notation
257
+ # writes them, and try again. If it parses, the space was
258
+ # cutting a construct in two and that is what to say. If it
259
+ # fails the same way, the space is a red herring and the cause
260
+ # already named is the one to keep.
261
+ #
262
+ # It used to keep :bare_hyphen and relabel every other cause,
263
+ # so one error said two things: the message naming an unknown
264
+ # colour while the code and the hint talked about spaces. A
265
+ # caller acting on the code — which is what these are for —
266
+ # was sent to fix what was not wrong.
267
+ raise e unless space_is_the_cause?(token_r.join, parent)
219
268
 
220
269
  raise RSTError.new(e.message,
221
270
  code: :label_split,