xlsxrb 0.1.8 → 0.1.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,490 @@
1
+ # frozen_string_literal: true
2
+
3
+ # rbs_inline: enabled
4
+
5
+ module Xlsxrb
6
+ module Ooxml
7
+ # Pure-Ruby Compound File Binary (CFB / OLE Structured Storage) implementation for [MS-CFB] / [MS-OFFCRYPTO].
8
+ module Cfb
9
+ MAGIC = "\xD0\xCF\x11\xE0\xA1\xB1\x1A\xE1".b.freeze
10
+
11
+ FREESECT = 0xFFFFFFFF
12
+ ENDOFCHAIN = 0xFFFFFFFE
13
+ FATSECT = 0xFFFFFFFD
14
+ DIFATSECT = 0xFFFFFFFC
15
+ NOSTREAM = 0xFFFFFFFF
16
+
17
+ OBJ_UNKNOWN = 0x00
18
+ OBJ_STORAGE = 0x01
19
+ OBJ_STREAM = 0x02
20
+ OBJ_ROOT = 0x05
21
+
22
+ MINI_STREAM_CUTOFF = 4096
23
+ SECTOR_SIZE = 512
24
+ MINI_SECTOR_SIZE = 64
25
+
26
+ # Represents a directory entry in a Compound File.
27
+ class DirEntry
28
+ attr_accessor :name, :type, :color, :left_sibling_id, :right_sibling_id, :child_id, :clsid, :state_flags, :created_time, :modified_time, :start_sector, :size, :entry_id
29
+
30
+ def initialize(name: "", type: OBJ_UNKNOWN, start_sector: ENDOFCHAIN, size: 0)
31
+ @name = name
32
+ @type = type
33
+ @color = 0 # Red / Black (0 is acceptable across OLE implementations)
34
+ @left_sibling_id = NOSTREAM
35
+ @right_sibling_id = NOSTREAM
36
+ @child_id = NOSTREAM
37
+ @clsid = "\x00".b * 16
38
+ @state_flags = 0
39
+ @created_time = 0
40
+ @modified_time = 0
41
+ @start_sector = start_sector
42
+ @size = size
43
+ @entry_id = 0
44
+ end
45
+
46
+ def stream?
47
+ @type == OBJ_STREAM
48
+ end
49
+
50
+ def root?
51
+ @type == OBJ_ROOT
52
+ end
53
+
54
+ def storage?
55
+ @type == OBJ_STORAGE
56
+ end
57
+ end
58
+
59
+ # Reads streams from a Compound File Binary buffer.
60
+ class Reader
61
+ attr_reader :entries
62
+
63
+ def self.cfb?(data)
64
+ return false if data.nil? || data.bytesize < 8
65
+
66
+ data[0, 8] == MAGIC
67
+ end
68
+
69
+ def initialize(data)
70
+ @data = data.b
71
+ raise Xlsxrb::Error, "Invalid CFB file header signature" unless Reader.cfb?(@data)
72
+
73
+ parse_header
74
+ build_fat
75
+ parse_directory
76
+ load_mini_stream
77
+ end
78
+
79
+ def stream_names
80
+ @entries.select(&:stream?).map(&:name)
81
+ end
82
+
83
+ def read_stream(name)
84
+ entry = @entries.find { |e| e.name.casecmp?(name) && e.stream? }
85
+ return nil unless entry
86
+
87
+ if entry.size < @mini_cutoff && @mini_stream && !@minifat.empty?
88
+ read_mini_stream_data(entry.start_sector, entry.size)
89
+ else
90
+ read_regular_stream_data(entry.start_sector, entry.size)
91
+ end
92
+ end
93
+
94
+ private
95
+
96
+ def parse_header
97
+ raise Xlsxrb::Error, "CFB file is too small" if @data.bytesize < 512
98
+
99
+ _, major_ver = @data[0x18, 4].unpack("v2")
100
+ @major_version = major_ver
101
+ @sector_shift = @data[0x1E, 2].unpack1("v")
102
+ @mini_sector_shift = @data[0x20, 2].unpack1("v")
103
+ @sector_size = 1 << @sector_shift
104
+ @mini_sector_size = 1 << @mini_sector_shift
105
+
106
+ @num_dir_sectors = @data[0x28, 4].unpack1("V")
107
+ @num_fat_sectors = @data[0x2C, 4].unpack1("V")
108
+ @first_dir_sector = @data[0x30, 4].unpack1("V")
109
+ @mini_cutoff = @data[0x38, 4].unpack1("V")
110
+ @first_minifat_sector = @data[0x3C, 4].unpack1("V")
111
+ @num_minifat_sectors = @data[0x40, 4].unpack1("V")
112
+ @first_difat_sector = @data[0x44, 4].unpack1("V")
113
+ @num_difat_sectors = @data[0x48, 4].unpack1("V")
114
+
115
+ @header_difat = @data[0x4C, 436].unpack("V109").reject { |s| [FREESECT, ENDOFCHAIN].include?(s) }
116
+ end
117
+
118
+ def sector_offset(sector_id)
119
+ 512 + (sector_id * @sector_size)
120
+ end
121
+
122
+ def read_sector(sector_id)
123
+ offset = sector_offset(sector_id)
124
+ @data[offset, @sector_size] || "".b
125
+ end
126
+
127
+ def build_fat
128
+ load_difat_and_fat
129
+ end
130
+
131
+ def load_difat_and_fat
132
+ @difat_sectors = @header_difat.dup
133
+ if @first_difat_sector != ENDOFCHAIN && @first_difat_sector != FREESECT
134
+ curr = @first_difat_sector
135
+ visited = {}
136
+ while curr != ENDOFCHAIN && curr != FREESECT
137
+ break if visited[curr]
138
+
139
+ visited[curr] = true
140
+ sec_data = read_sector(curr)
141
+ entries_per_sec = (@sector_size / 4) - 1
142
+ entries = sec_data[0, entries_per_sec * 4].unpack("V*")
143
+ @difat_sectors.concat(entries)
144
+ curr = sec_data[entries_per_sec * 4, 4].unpack1("V")
145
+ end
146
+ end
147
+
148
+ # Read all FAT sectors
149
+ @fat = []
150
+ @difat_sectors.each do |fat_sec_id|
151
+ break if [ENDOFCHAIN, FREESECT].include?(fat_sec_id)
152
+
153
+ sec_data = read_sector(fat_sec_id)
154
+ @fat.concat(sec_data.unpack("V*"))
155
+ end
156
+ end
157
+
158
+ def parse_directory
159
+ dir_data = +""
160
+ curr = @first_dir_sector
161
+ visited = {}
162
+ while curr != ENDOFCHAIN && curr != FREESECT && curr < @fat.size
163
+ break if visited[curr]
164
+
165
+ visited[curr] = true
166
+ dir_data << read_sector(curr)
167
+ curr = @fat[curr]
168
+ end
169
+
170
+ @entries = []
171
+ num_entries = dir_data.bytesize / 128
172
+ num_entries.times do |i|
173
+ entry_bytes = dir_data[i * 128, 128]
174
+ next if entry_bytes.nil? || entry_bytes.bytesize < 128
175
+
176
+ name_bytes = entry_bytes[0, 64]
177
+ name_len = entry_bytes[0x40, 2].unpack1("v")
178
+ type = entry_bytes[0x42, 1].ord
179
+ next if type == OBJ_UNKNOWN
180
+
181
+ name_str = if name_len > 2
182
+ name_bytes[0, name_len - 2].force_encoding("UTF-16LE").encode("UTF-8", invalid: :replace, undef: :replace)
183
+ else
184
+ ""
185
+ end
186
+
187
+ entry = DirEntry.new(name: name_str, type: type)
188
+ entry.entry_id = i
189
+ entry.color = entry_bytes[0x43, 1].ord
190
+ entry.left_sibling_id = entry_bytes[0x44, 4].unpack1("V")
191
+ entry.right_sibling_id = entry_bytes[0x48, 4].unpack1("V")
192
+ entry.child_id = entry_bytes[0x4C, 4].unpack1("V")
193
+ entry.clsid = entry_bytes[0x50, 16]
194
+ entry.state_flags = entry_bytes[0x60, 4].unpack1("V")
195
+ entry.start_sector = entry_bytes[0x74, 4].unpack1("V")
196
+ entry.size = entry_bytes[0x78, 8].unpack1("Q<")
197
+
198
+ @entries << entry
199
+ end
200
+ end
201
+
202
+ def load_mini_stream
203
+ @minifat = []
204
+ if @first_minifat_sector != ENDOFCHAIN && @first_minifat_sector != FREESECT
205
+ curr = @first_minifat_sector
206
+ visited = {}
207
+ while curr != ENDOFCHAIN && curr != FREESECT && curr < @fat.size
208
+ break if visited[curr]
209
+
210
+ visited[curr] = true
211
+ sec_data = read_sector(curr)
212
+ @minifat.concat(sec_data.unpack("V*"))
213
+ curr = @fat[curr]
214
+ end
215
+ end
216
+
217
+ root_entry = @entries.find(&:root?)
218
+ @mini_stream = if root_entry && root_entry.start_sector != ENDOFCHAIN && root_entry.size.positive?
219
+ read_regular_stream_data(root_entry.start_sector, root_entry.size)
220
+ else
221
+ "".b
222
+ end
223
+ end
224
+
225
+ def read_regular_stream_data(start_sector, total_size)
226
+ return "".b if [ENDOFCHAIN, FREESECT].include?(start_sector) || total_size.zero?
227
+
228
+ result = +""
229
+ curr = start_sector
230
+ visited = {}
231
+ while curr != ENDOFCHAIN && curr != FREESECT && curr < @fat.size
232
+ break if visited[curr]
233
+
234
+ visited[curr] = true
235
+ result << read_sector(curr)
236
+ break if result.bytesize >= total_size
237
+
238
+ curr = @fat[curr]
239
+ end
240
+ result[0, total_size] || "".b
241
+ end
242
+
243
+ def read_mini_stream_data(start_mini_sector, total_size)
244
+ return "".b if [ENDOFCHAIN, FREESECT].include?(start_mini_sector) || total_size.zero?
245
+
246
+ result = +""
247
+ curr = start_mini_sector
248
+ visited = {}
249
+ while curr != ENDOFCHAIN && curr != FREESECT && curr < @minifat.size
250
+ break if visited[curr]
251
+
252
+ visited[curr] = true
253
+ offset = curr * @mini_sector_size
254
+ result << (@mini_stream[offset, @mini_sector_size] || "".b)
255
+ break if result.bytesize >= total_size
256
+
257
+ curr = @minifat[curr]
258
+ end
259
+ result[0, total_size] || "".b
260
+ end
261
+ end
262
+
263
+ # Writes named streams into a Compound File Binary (v3, 512-byte sectors) format with Mini Stream support.
264
+ class Writer
265
+ def self.write(streams)
266
+ new(streams).build
267
+ end
268
+
269
+ def initialize(streams)
270
+ # streams: Hash of { String => String (bytes) }
271
+ @streams = streams
272
+ end
273
+
274
+ def build
275
+ sector_size = SECTOR_SIZE
276
+ mini_sector_size = MINI_SECTOR_SIZE
277
+
278
+ # Partition streams into mini streams (< 4096 bytes) and regular streams (>= 4096 bytes)
279
+ regular_stream_entries = []
280
+
281
+ mini_stream_bytes = +""
282
+ minifat = []
283
+
284
+ # Directory Entries array
285
+ dir_entries = []
286
+ root_entry = DirEntry.new(name: "Root Entry", type: OBJ_ROOT, start_sector: ENDOFCHAIN, size: 0)
287
+ dir_entries << root_entry
288
+
289
+ stream_names = @streams.keys
290
+ stream_names.each_with_index do |name, idx|
291
+ entry_id = idx + 1
292
+ data = @streams[name].b
293
+ size = data.bytesize
294
+
295
+ if size < MINI_STREAM_CUTOFF && size.positive?
296
+ # Allocate in Mini Stream
297
+ start_mini_sec = minifat.size
298
+ num_mini_sec = (size + mini_sector_size - 1) / mini_sector_size
299
+ num_mini_sec.times do |m_idx|
300
+ chunk = data[m_idx * mini_sector_size, mini_sector_size] || "".b
301
+ chunk = chunk.ljust(mini_sector_size, "\x00".b) if chunk.bytesize < mini_sector_size
302
+ mini_stream_bytes << chunk
303
+ minifat << (m_idx == num_mini_sec - 1 ? ENDOFCHAIN : (start_mini_sec + m_idx + 1))
304
+ end
305
+ entry = DirEntry.new(name: name, type: OBJ_STREAM, start_sector: start_mini_sec, size: size)
306
+ elsif size >= MINI_STREAM_CUTOFF
307
+ # Allocate in Regular Stream (start_sector will be assigned later)
308
+ entry = DirEntry.new(name: name, type: OBJ_STREAM, start_sector: ENDOFCHAIN, size: size)
309
+ regular_stream_entries << [entry, data]
310
+ else
311
+ entry = DirEntry.new(name: name, type: OBJ_STREAM, start_sector: ENDOFCHAIN, size: 0)
312
+ end
313
+
314
+ entry.entry_id = entry_id
315
+ dir_entries << entry
316
+ end
317
+
318
+ # Set up directory binary tree
319
+ root_entry.child_id = @streams.empty? ? NOSTREAM : 1
320
+ if dir_entries.size > 1
321
+ root_child = dir_entries[1]
322
+ (2...dir_entries.size).each do |i|
323
+ insert_entry_to_tree(dir_entries, root_child, dir_entries[i])
324
+ end
325
+ end
326
+
327
+ # Pad directory entries to multiple of 4 (128 bytes * 4 = 512 bytes = 1 sector)
328
+ dir_entries << DirEntry.new(type: OBJ_UNKNOWN) until (dir_entries.size % 4).zero?
329
+
330
+ # Now allocate regular sectors:
331
+ # 1. Regular stream data sectors
332
+ # 2. Mini stream container sectors
333
+ # 3. Mini FAT sectors
334
+ # 4. Directory sector
335
+ # 5. FAT sector
336
+ allocated_sectors = []
337
+ sector_chains = [] # pairs of [start_sec, num_sec]
338
+
339
+ # 1. Regular streams
340
+ regular_stream_entries.each do |entry, data|
341
+ start_sec = allocated_sectors.size
342
+ num_sec = (data.bytesize + sector_size - 1) / sector_size
343
+ num_sec.times do |s_idx|
344
+ chunk = data[s_idx * sector_size, sector_size] || "".b
345
+ chunk = chunk.ljust(sector_size, "\x00".b) if chunk.bytesize < sector_size
346
+ allocated_sectors << chunk
347
+ end
348
+ entry.start_sector = start_sec
349
+ sector_chains << [start_sec, num_sec]
350
+ end
351
+
352
+ # 2. Mini Stream container (assigned to Root Entry)
353
+ if mini_stream_bytes.bytesize.positive?
354
+ start_sec = allocated_sectors.size
355
+ num_sec = (mini_stream_bytes.bytesize + sector_size - 1) / sector_size
356
+ num_sec.times do |s_idx|
357
+ chunk = mini_stream_bytes[s_idx * sector_size, sector_size] || "".b
358
+ chunk = chunk.ljust(sector_size, "\x00".b) if chunk.bytesize < sector_size
359
+ allocated_sectors << chunk
360
+ end
361
+ root_entry.start_sector = start_sec
362
+ root_entry.size = mini_stream_bytes.bytesize
363
+ sector_chains << [start_sec, num_sec]
364
+ end
365
+
366
+ # 3. Mini FAT sectors
367
+ first_minifat_sec = ENDOFCHAIN
368
+ num_minifat_sec = 0
369
+ if minifat.size.positive?
370
+ first_minifat_sec = allocated_sectors.size
371
+ minifat_bytes = minifat.pack("V*")
372
+ num_minifat_sec = (minifat_bytes.bytesize + sector_size - 1) / sector_size
373
+ num_minifat_sec.times do |s_idx|
374
+ chunk = minifat_bytes[s_idx * sector_size, sector_size] || "".b
375
+ chunk = chunk.ljust(sector_size, "\xFF".b) if chunk.bytesize < sector_size
376
+ allocated_sectors << chunk
377
+ end
378
+ sector_chains << [first_minifat_sec, num_minifat_sec]
379
+ end
380
+
381
+ # 4. Directory sector
382
+ dir_sector_id = allocated_sectors.size
383
+ dir_sector_bytes = +""
384
+ dir_entries.each do |e|
385
+ dir_sector_bytes << serialize_dir_entry(e)
386
+ end
387
+ allocated_sectors << dir_sector_bytes
388
+
389
+ # 5. FAT sector
390
+ fat_sector_id = allocated_sectors.size
391
+ fat = Array.new(fat_sector_id + 1, FREESECT)
392
+
393
+ # Populate regular sector chains in FAT
394
+ sector_chains.each do |start_sec, num_sec|
395
+ num_sec.times do |s_idx|
396
+ fat[start_sec + s_idx] = s_idx == num_sec - 1 ? ENDOFCHAIN : (start_sec + s_idx + 1)
397
+ end
398
+ end
399
+
400
+ fat[dir_sector_id] = ENDOFCHAIN
401
+ fat[fat_sector_id] = FATSECT
402
+
403
+ fat_bytes = fat.pack("V*").ljust(sector_size, "\xFF".b)
404
+ allocated_sectors << fat_bytes
405
+
406
+ # Build Header (512 bytes)
407
+ header = +""
408
+ header << MAGIC # 0x00 (8 bytes)
409
+ header << ("\x00".b * 16) # 0x08 CLSID
410
+ header << [0x003B, 0x0003].pack("v2") # 0x18 Minor version (0x3B), Major version (3)
411
+ header << [0xFFFE].pack("v") # 0x1C Byte order (Little Endian)
412
+ header << [9].pack("v") # 0x1E Sector shift (512 bytes)
413
+ header << [6].pack("v") # 0x20 Mini sector shift (64 bytes)
414
+ header << ("\x00".b * 6) # 0x22 Reserved
415
+ header << [0].pack("V") # 0x28 Number of Directory sectors (0 for v3)
416
+ header << [1].pack("V") # 0x2C Number of FAT sectors (1)
417
+ header << [dir_sector_id].pack("V") # 0x30 First Directory sector
418
+ header << [0].pack("V") # 0x34 Transaction signature
419
+ header << [MINI_STREAM_CUTOFF].pack("V") # 0x38 Mini stream cutoff size (4096)
420
+ header << [first_minifat_sec].pack("V") # 0x3C First Mini FAT sector
421
+ header << [num_minifat_sec].pack("V") # 0x40 Number of Mini FAT sectors
422
+ header << [ENDOFCHAIN].pack("V") # 0x44 First DIFAT sector
423
+ header << [0].pack("V") # 0x48 Number of DIFAT sectors
424
+
425
+ # DIFAT array (109 entries * 4 bytes = 436 bytes)
426
+ difat = Array.new(109, FREESECT)
427
+ difat[0] = fat_sector_id
428
+ header << difat.pack("V109")
429
+
430
+ raise "Header size mismatch" unless header.bytesize == 512
431
+
432
+ # Combine Header + All Sectors
433
+ output = +""
434
+ output << header
435
+ allocated_sectors.each { |sec| output << sec }
436
+ output
437
+ end
438
+
439
+ private
440
+
441
+ def serialize_dir_entry(entry)
442
+ return "\x00".b * 128 if entry.type == OBJ_UNKNOWN
443
+
444
+ buf = +""
445
+ # Name in UTF-16LE with null terminator
446
+ name_utf16 = "#{entry.name}\u0000".encode("UTF-16LE")
447
+ name_bytes = name_utf16.b[0, 64].ljust(64, "\x00".b)
448
+ buf << name_bytes
449
+ buf << [name_utf16.bytesize].pack("v") # 0x40 Name length
450
+ buf << [entry.type].pack("C") # 0x42 Object type
451
+ buf << [entry.color].pack("C") # 0x43 Color (0 = Red/Black)
452
+ buf << [entry.left_sibling_id].pack("V") # 0x44 Left sibling
453
+ buf << [entry.right_sibling_id].pack("V") # 0x48 Right sibling
454
+ buf << [entry.child_id].pack("V") # 0x4C Child ID
455
+ buf << (entry.clsid || ("\x00".b * 16)) # 0x50 CLSID
456
+ buf << [entry.state_flags].pack("V") # 0x60 State flags
457
+ buf << [entry.created_time].pack("Q<") # 0x64 Created time
458
+ buf << [entry.modified_time].pack("Q<") # 0x6C Modified time
459
+ buf << [entry.start_sector].pack("V") # 0x74 Starting sector
460
+ buf << [entry.size].pack("Q<") # 0x78 Stream size (uint64)
461
+
462
+ buf.ljust(128, "\x00".b)
463
+ end
464
+
465
+ def insert_entry_to_tree(entries, root_node, new_node)
466
+ cmp = compare_entry_names(new_node.name, root_node.name)
467
+ if cmp.negative?
468
+ if root_node.left_sibling_id == NOSTREAM
469
+ root_node.left_sibling_id = new_node.entry_id
470
+ else
471
+ insert_entry_to_tree(entries, entries[root_node.left_sibling_id], new_node)
472
+ end
473
+ elsif root_node.right_sibling_id == NOSTREAM
474
+ root_node.right_sibling_id = new_node.entry_id
475
+ else
476
+ insert_entry_to_tree(entries, entries[root_node.right_sibling_id], new_node)
477
+ end
478
+ end
479
+
480
+ def compare_entry_names(a_name, b_name)
481
+ # [MS-CFB] Section 2.6.1: Length comparison first, then uppercase UTF-16 code point comparison
482
+ return -1 if a_name.length < b_name.length
483
+ return 1 if a_name.length > b_name.length
484
+
485
+ a_name.upcase <=> b_name.upcase
486
+ end
487
+ end
488
+ end
489
+ end
490
+ end