extract-pdf 0.1.21 → 0.1.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -74
- package/package.json +1 -1
- package/src/models/annotation.ts +41 -41
- package/src/models/block-type.ts +203 -203
- package/src/models/line-converter.ts +224 -224
- package/src/models/metadata.ts +29 -29
- package/src/models/page.ts +16 -16
- package/src/models/parse-result.ts +32 -32
- package/src/models/parsed-elements.ts +29 -29
- package/src/models/stashing-stream.ts +86 -86
- package/src/models/text-item-line-grouper.ts +41 -41
- package/src/models/word.ts +31 -31
- package/src/pdf-to-html.ts +225 -225
- package/src/transforms/base/to-line-item-block-transform.ts +29 -29
- package/src/transforms/base/to-line-item-transform.ts +29 -29
- package/src/transforms/base/to-text-item-transform.ts +28 -28
- package/src/transforms/block/detect-code-quote-blocks.ts +57 -57
- package/src/transforms/block/detect-list-levels.ts +64 -64
- package/src/transforms/block/gather-blocks.ts +113 -113
- package/src/transforms/calculate-global-stats.ts +132 -132
- package/src/transforms/line-item/compact-lines.ts +92 -92
- package/src/transforms/line-item/detect-headers.ts +173 -173
- package/src/transforms/line-item/detect-list-items.ts +68 -68
- package/src/transforms/line-item/detect-toc.ts +459 -459
- package/src/transforms/line-item/remove-repetitive-elements.ts +101 -101
- package/src/transforms/line-item/vertical-to-horizontal.ts +90 -90
- package/src/transforms/to-html.ts +46 -46
- package/src/utils/is-url-pdf.ts +33 -33
- package/src/utils/page-item-functions.ts +35 -35
- package/src/utils/string-functions.ts +124 -124
|
@@ -1,57 +1,57 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Block-level transformation that marks blocks as `BlockType.CODE`
|
|
3
|
-
* when all constituent lines are indented beyond the page's leftmost x position
|
|
4
|
-
* and the text height matches body-text height. Heuristically distinguishes
|
|
5
|
-
* indented code/blockquote sections from regular paragraphs without relying on
|
|
6
|
-
* font metadata.
|
|
7
|
-
*/
|
|
8
|
-
|
|
9
|
-
import ToLineItemBlockTransformation from '../base/to-line-item-block-transform'
|
|
10
|
-
import ParseResult from '../../models/parse-result'
|
|
11
|
-
import { DETECTED_ANNOTATION } from '../../models/annotation'
|
|
12
|
-
import BlockType from '../../models/block-type'
|
|
13
|
-
import { minXFromBlocks } from '../../utils/page-item-functions'
|
|
14
|
-
|
|
15
|
-
// Detect items which are code/quote blocks
|
|
16
|
-
export default class DetectCodeQuoteBlocks extends ToLineItemBlockTransformation {
|
|
17
|
-
constructor () {
|
|
18
|
-
super('$1')
|
|
19
|
-
}
|
|
20
|
-
|
|
21
|
-
transform(parseResult: ParseResult): ParseResult {
|
|
22
|
-
const mostUsedHeight = parseResult.globals.mostUsedHeight ?? 0
|
|
23
|
-
var foundCodeItems = 0
|
|
24
|
-
parseResult.pages.forEach(page => {
|
|
25
|
-
var minX = minXFromBlocks(page.items)
|
|
26
|
-
page.items.forEach(block => {
|
|
27
|
-
if (!block.type && looksLikeCodeBlock(minX, block.items, mostUsedHeight)) {
|
|
28
|
-
block.annotation = DETECTED_ANNOTATION
|
|
29
|
-
block.type = BlockType.CODE
|
|
30
|
-
foundCodeItems++
|
|
31
|
-
}
|
|
32
|
-
})
|
|
33
|
-
})
|
|
34
|
-
|
|
35
|
-
return new ParseResult({
|
|
36
|
-
...parseResult,
|
|
37
|
-
messages: [
|
|
38
|
-
'Detected ' + foundCodeItems + ' code/quote items.',
|
|
39
|
-
],
|
|
40
|
-
})
|
|
41
|
-
}
|
|
42
|
-
}
|
|
43
|
-
|
|
44
|
-
function looksLikeCodeBlock(minX: number, items: any[], mostUsedHeight: number): boolean {
|
|
45
|
-
if (items.length === 0) {
|
|
46
|
-
return false
|
|
47
|
-
}
|
|
48
|
-
if (items.length === 1) {
|
|
49
|
-
return items[0].x > minX && items[0].height <= mostUsedHeight + 1
|
|
50
|
-
}
|
|
51
|
-
for (var item of items) {
|
|
52
|
-
if (item.x === minX) {
|
|
53
|
-
return false
|
|
54
|
-
}
|
|
55
|
-
}
|
|
56
|
-
return true
|
|
57
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @description Block-level transformation that marks blocks as `BlockType.CODE`
|
|
3
|
+
* when all constituent lines are indented beyond the page's leftmost x position
|
|
4
|
+
* and the text height matches body-text height. Heuristically distinguishes
|
|
5
|
+
* indented code/blockquote sections from regular paragraphs without relying on
|
|
6
|
+
* font metadata.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import ToLineItemBlockTransformation from '../base/to-line-item-block-transform'
|
|
10
|
+
import ParseResult from '../../models/parse-result'
|
|
11
|
+
import { DETECTED_ANNOTATION } from '../../models/annotation'
|
|
12
|
+
import BlockType from '../../models/block-type'
|
|
13
|
+
import { minXFromBlocks } from '../../utils/page-item-functions'
|
|
14
|
+
|
|
15
|
+
// Detect items which are code/quote blocks
|
|
16
|
+
export default class DetectCodeQuoteBlocks extends ToLineItemBlockTransformation {
|
|
17
|
+
constructor () {
|
|
18
|
+
super('$1')
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
22
|
+
const mostUsedHeight = parseResult.globals.mostUsedHeight ?? 0
|
|
23
|
+
var foundCodeItems = 0
|
|
24
|
+
parseResult.pages.forEach(page => {
|
|
25
|
+
var minX = minXFromBlocks(page.items)
|
|
26
|
+
page.items.forEach(block => {
|
|
27
|
+
if (!block.type && looksLikeCodeBlock(minX, block.items, mostUsedHeight)) {
|
|
28
|
+
block.annotation = DETECTED_ANNOTATION
|
|
29
|
+
block.type = BlockType.CODE
|
|
30
|
+
foundCodeItems++
|
|
31
|
+
}
|
|
32
|
+
})
|
|
33
|
+
})
|
|
34
|
+
|
|
35
|
+
return new ParseResult({
|
|
36
|
+
...parseResult,
|
|
37
|
+
messages: [
|
|
38
|
+
'Detected ' + foundCodeItems + ' code/quote items.',
|
|
39
|
+
],
|
|
40
|
+
})
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function looksLikeCodeBlock(minX: number, items: any[], mostUsedHeight: number): boolean {
|
|
45
|
+
if (items.length === 0) {
|
|
46
|
+
return false
|
|
47
|
+
}
|
|
48
|
+
if (items.length === 1) {
|
|
49
|
+
return items[0].x > minX && items[0].height <= mostUsedHeight + 1
|
|
50
|
+
}
|
|
51
|
+
for (var item of items) {
|
|
52
|
+
if (item.x === minX) {
|
|
53
|
+
return false
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
return true
|
|
57
|
+
}
|
|
@@ -1,64 +1,64 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Block-level transformation that adds hierarchical indentation to
|
|
3
|
-
* nested list items within `BlockType.LIST` blocks. Tracks nesting level by
|
|
4
|
-
* comparing each item's x position to the previous item's — deeper x means a
|
|
5
|
-
* new sub-level — and prepends spaces proportional to the level so the HTML
|
|
6
|
-
* output reflects the original visual hierarchy.
|
|
7
|
-
*/
|
|
8
|
-
import ToLineItemBlockTransformation from '../base/to-line-item-block-transform'
|
|
9
|
-
import ParseResult from '../../models/parse-result'
|
|
10
|
-
import Word from '../../models/word'
|
|
11
|
-
import { MODIFIED_ANNOTATION, UNCHANGED_ANNOTATION } from '../../models/annotation'
|
|
12
|
-
import BlockType from '../../models/block-type'
|
|
13
|
-
|
|
14
|
-
// Cares for proper sub-item spacing/leveling
|
|
15
|
-
export default class DetectListLevels extends ToLineItemBlockTransformation {
|
|
16
|
-
constructor () {
|
|
17
|
-
super('Level Lists')
|
|
18
|
-
}
|
|
19
|
-
|
|
20
|
-
transform(parseResult: ParseResult): ParseResult {
|
|
21
|
-
var listBlocks = 0
|
|
22
|
-
var modifiedBlocks = 0
|
|
23
|
-
parseResult.pages.forEach(page => {
|
|
24
|
-
page.items.filter(block => block.type === BlockType.LIST).forEach(listBlock => {
|
|
25
|
-
var lastItemX: number | undefined
|
|
26
|
-
var currentLevel = 0
|
|
27
|
-
const xByLevel: Record<number, number> = {}
|
|
28
|
-
var modifiedBlock = false
|
|
29
|
-
listBlock.items.forEach(item => {
|
|
30
|
-
const isListItem = true
|
|
31
|
-
if (lastItemX && isListItem) {
|
|
32
|
-
if (item.x > lastItemX) {
|
|
33
|
-
currentLevel++
|
|
34
|
-
xByLevel[item.x] = currentLevel
|
|
35
|
-
} else if (item.x < lastItemX) {
|
|
36
|
-
currentLevel = xByLevel[item.x]
|
|
37
|
-
}
|
|
38
|
-
} else {
|
|
39
|
-
xByLevel[item.x] = 0
|
|
40
|
-
}
|
|
41
|
-
if (currentLevel > 0) {
|
|
42
|
-
item.words = [
|
|
43
|
-
new Word({ string: ' '.repeat(currentLevel * 3) }),
|
|
44
|
-
].concat(item.words)
|
|
45
|
-
modifiedBlock = true
|
|
46
|
-
}
|
|
47
|
-
lastItemX = item.x
|
|
48
|
-
})
|
|
49
|
-
listBlocks++
|
|
50
|
-
if (modifiedBlock) {
|
|
51
|
-
modifiedBlocks++
|
|
52
|
-
listBlock.annotation = MODIFIED_ANNOTATION
|
|
53
|
-
} else {
|
|
54
|
-
listBlock.annotation = UNCHANGED_ANNOTATION
|
|
55
|
-
}
|
|
56
|
-
})
|
|
57
|
-
})
|
|
58
|
-
|
|
59
|
-
return new ParseResult({
|
|
60
|
-
...parseResult,
|
|
61
|
-
messages: ['Modified ' + modifiedBlocks + ' / ' + listBlocks + ' list blocks.'],
|
|
62
|
-
})
|
|
63
|
-
}
|
|
64
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @description Block-level transformation that adds hierarchical indentation to
|
|
3
|
+
* nested list items within `BlockType.LIST` blocks. Tracks nesting level by
|
|
4
|
+
* comparing each item's x position to the previous item's — deeper x means a
|
|
5
|
+
* new sub-level — and prepends spaces proportional to the level so the HTML
|
|
6
|
+
* output reflects the original visual hierarchy.
|
|
7
|
+
*/
|
|
8
|
+
import ToLineItemBlockTransformation from '../base/to-line-item-block-transform'
|
|
9
|
+
import ParseResult from '../../models/parse-result'
|
|
10
|
+
import Word from '../../models/word'
|
|
11
|
+
import { MODIFIED_ANNOTATION, UNCHANGED_ANNOTATION } from '../../models/annotation'
|
|
12
|
+
import BlockType from '../../models/block-type'
|
|
13
|
+
|
|
14
|
+
// Cares for proper sub-item spacing/leveling
|
|
15
|
+
export default class DetectListLevels extends ToLineItemBlockTransformation {
|
|
16
|
+
constructor () {
|
|
17
|
+
super('Level Lists')
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
21
|
+
var listBlocks = 0
|
|
22
|
+
var modifiedBlocks = 0
|
|
23
|
+
parseResult.pages.forEach(page => {
|
|
24
|
+
page.items.filter(block => block.type === BlockType.LIST).forEach(listBlock => {
|
|
25
|
+
var lastItemX: number | undefined
|
|
26
|
+
var currentLevel = 0
|
|
27
|
+
const xByLevel: Record<number, number> = {}
|
|
28
|
+
var modifiedBlock = false
|
|
29
|
+
listBlock.items.forEach(item => {
|
|
30
|
+
const isListItem = true
|
|
31
|
+
if (lastItemX && isListItem) {
|
|
32
|
+
if (item.x > lastItemX) {
|
|
33
|
+
currentLevel++
|
|
34
|
+
xByLevel[item.x] = currentLevel
|
|
35
|
+
} else if (item.x < lastItemX) {
|
|
36
|
+
currentLevel = xByLevel[item.x]
|
|
37
|
+
}
|
|
38
|
+
} else {
|
|
39
|
+
xByLevel[item.x] = 0
|
|
40
|
+
}
|
|
41
|
+
if (currentLevel > 0) {
|
|
42
|
+
item.words = [
|
|
43
|
+
new Word({ string: ' '.repeat(currentLevel * 3) }),
|
|
44
|
+
].concat(item.words)
|
|
45
|
+
modifiedBlock = true
|
|
46
|
+
}
|
|
47
|
+
lastItemX = item.x
|
|
48
|
+
})
|
|
49
|
+
listBlocks++
|
|
50
|
+
if (modifiedBlock) {
|
|
51
|
+
modifiedBlocks++
|
|
52
|
+
listBlock.annotation = MODIFIED_ANNOTATION
|
|
53
|
+
} else {
|
|
54
|
+
listBlock.annotation = UNCHANGED_ANNOTATION
|
|
55
|
+
}
|
|
56
|
+
})
|
|
57
|
+
})
|
|
58
|
+
|
|
59
|
+
return new ParseResult({
|
|
60
|
+
...parseResult,
|
|
61
|
+
messages: ['Modified ' + modifiedBlocks + ' / ' + listBlocks + ' list blocks.'],
|
|
62
|
+
})
|
|
63
|
+
}
|
|
64
|
+
}
|
|
@@ -1,113 +1,113 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @description Block-level transformation that merges consecutive `LineItem`s into
|
|
3
|
-
* `LineItemBlock`s. Flushes the current block when the item type changes or the
|
|
4
|
-
* vertical gap exceeds `mostUsedDistance`, while respecting per-type merge flags
|
|
5
|
-
* (`mergeToBlock`, `mergeFollowingNonTypedItems`,
|
|
6
|
-
* `mergeFollowingNonTypedItemsWithSmallDistance`) to handle lists and footnotes correctly.
|
|
7
|
-
*/
|
|
8
|
-
|
|
9
|
-
import ToLineItemBlockTransformation from "../base/to-line-item-block-transform";
|
|
10
|
-
import ParseResult from "../../models/parse-result";
|
|
11
|
-
import LineItemBlock from "../../models/line-item-block";
|
|
12
|
-
import { DETECTED_ANNOTATION } from "../../models/annotation";
|
|
13
|
-
import { minXFromPageItems } from "../../utils/page-item-functions";
|
|
14
|
-
|
|
15
|
-
// Gathers lines to blocks
|
|
16
|
-
export default class GatherBlocks extends ToLineItemBlockTransformation {
|
|
17
|
-
constructor() {
|
|
18
|
-
super("Gather Blocks");
|
|
19
|
-
}
|
|
20
|
-
|
|
21
|
-
transform(parseResult: ParseResult): ParseResult {
|
|
22
|
-
const { mostUsedDistance } = parseResult.globals;
|
|
23
|
-
var createdBlocks = 0;
|
|
24
|
-
var lineItemCount = 0;
|
|
25
|
-
parseResult.pages.map((page) => {
|
|
26
|
-
lineItemCount += page.items.length;
|
|
27
|
-
const blocks = [];
|
|
28
|
-
var stashedBlock = new LineItemBlock({});
|
|
29
|
-
const flushStashedItems = () => {
|
|
30
|
-
if (stashedBlock.items.length > 1) {
|
|
31
|
-
stashedBlock.annotation = DETECTED_ANNOTATION;
|
|
32
|
-
}
|
|
33
|
-
|
|
34
|
-
blocks.push(stashedBlock);
|
|
35
|
-
stashedBlock = new LineItemBlock({});
|
|
36
|
-
createdBlocks++;
|
|
37
|
-
};
|
|
38
|
-
|
|
39
|
-
var minX = minXFromPageItems(page.items);
|
|
40
|
-
page.items.forEach((item) => {
|
|
41
|
-
if (
|
|
42
|
-
stashedBlock.items.length > 0 &&
|
|
43
|
-
shouldFlushBlock(stashedBlock, item, minX, mostUsedDistance)
|
|
44
|
-
) {
|
|
45
|
-
flushStashedItems();
|
|
46
|
-
}
|
|
47
|
-
stashedBlock.addItem(item);
|
|
48
|
-
});
|
|
49
|
-
if (stashedBlock.items.length > 0) {
|
|
50
|
-
flushStashedItems();
|
|
51
|
-
}
|
|
52
|
-
page.items = blocks;
|
|
53
|
-
});
|
|
54
|
-
|
|
55
|
-
return new ParseResult({
|
|
56
|
-
...parseResult,
|
|
57
|
-
messages: [
|
|
58
|
-
"Gathered " +
|
|
59
|
-
createdBlocks +
|
|
60
|
-
" blocks out of " +
|
|
61
|
-
lineItemCount +
|
|
62
|
-
" line items",
|
|
63
|
-
],
|
|
64
|
-
});
|
|
65
|
-
}
|
|
66
|
-
}
|
|
67
|
-
|
|
68
|
-
function shouldFlushBlock(stashedBlock: any, item: any, minX: number, mostUsedDistance: number): boolean {
|
|
69
|
-
if (
|
|
70
|
-
stashedBlock.type &&
|
|
71
|
-
stashedBlock.type.mergeFollowingNonTypedItems &&
|
|
72
|
-
!item.type
|
|
73
|
-
) {
|
|
74
|
-
return false;
|
|
75
|
-
}
|
|
76
|
-
const lastItem = stashedBlock.items[stashedBlock.items.length - 1];
|
|
77
|
-
const hasBigDistance = bigDistance(lastItem, item, minX, mostUsedDistance);
|
|
78
|
-
if (
|
|
79
|
-
stashedBlock.type &&
|
|
80
|
-
stashedBlock.type.mergeFollowingNonTypedItemsWithSmallDistance &&
|
|
81
|
-
!item.type &&
|
|
82
|
-
!hasBigDistance
|
|
83
|
-
) {
|
|
84
|
-
return false;
|
|
85
|
-
}
|
|
86
|
-
if (item.type !== stashedBlock.type) {
|
|
87
|
-
return true;
|
|
88
|
-
}
|
|
89
|
-
if (item.type) {
|
|
90
|
-
return !item.type.mergeToBlock;
|
|
91
|
-
} else {
|
|
92
|
-
return hasBigDistance;
|
|
93
|
-
}
|
|
94
|
-
}
|
|
95
|
-
|
|
96
|
-
function bigDistance(lastItem: any, item: any, minX: number, mostUsedDistance: number): boolean {
|
|
97
|
-
const distance = lastItem.y - item.y;
|
|
98
|
-
if (distance < 0 - mostUsedDistance / 2) {
|
|
99
|
-
// distance is negative - and not only a bit
|
|
100
|
-
return true;
|
|
101
|
-
}
|
|
102
|
-
var allowedDisctance = mostUsedDistance + 1;
|
|
103
|
-
if (lastItem.x > minX && item.x > minX) {
|
|
104
|
-
// intended elements like lists often have greater spacing
|
|
105
|
-
allowedDisctance = mostUsedDistance + mostUsedDistance / 2;
|
|
106
|
-
}
|
|
107
|
-
if (distance > allowedDisctance) {
|
|
108
|
-
return true;
|
|
109
|
-
}
|
|
110
|
-
return false;
|
|
111
|
-
}
|
|
112
|
-
|
|
113
|
-
|
|
1
|
+
/**
|
|
2
|
+
* @description Block-level transformation that merges consecutive `LineItem`s into
|
|
3
|
+
* `LineItemBlock`s. Flushes the current block when the item type changes or the
|
|
4
|
+
* vertical gap exceeds `mostUsedDistance`, while respecting per-type merge flags
|
|
5
|
+
* (`mergeToBlock`, `mergeFollowingNonTypedItems`,
|
|
6
|
+
* `mergeFollowingNonTypedItemsWithSmallDistance`) to handle lists and footnotes correctly.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import ToLineItemBlockTransformation from "../base/to-line-item-block-transform";
|
|
10
|
+
import ParseResult from "../../models/parse-result";
|
|
11
|
+
import LineItemBlock from "../../models/line-item-block";
|
|
12
|
+
import { DETECTED_ANNOTATION } from "../../models/annotation";
|
|
13
|
+
import { minXFromPageItems } from "../../utils/page-item-functions";
|
|
14
|
+
|
|
15
|
+
// Gathers lines to blocks
|
|
16
|
+
export default class GatherBlocks extends ToLineItemBlockTransformation {
|
|
17
|
+
constructor() {
|
|
18
|
+
super("Gather Blocks");
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
transform(parseResult: ParseResult): ParseResult {
|
|
22
|
+
const { mostUsedDistance } = parseResult.globals;
|
|
23
|
+
var createdBlocks = 0;
|
|
24
|
+
var lineItemCount = 0;
|
|
25
|
+
parseResult.pages.map((page) => {
|
|
26
|
+
lineItemCount += page.items.length;
|
|
27
|
+
const blocks = [];
|
|
28
|
+
var stashedBlock = new LineItemBlock({});
|
|
29
|
+
const flushStashedItems = () => {
|
|
30
|
+
if (stashedBlock.items.length > 1) {
|
|
31
|
+
stashedBlock.annotation = DETECTED_ANNOTATION;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
blocks.push(stashedBlock);
|
|
35
|
+
stashedBlock = new LineItemBlock({});
|
|
36
|
+
createdBlocks++;
|
|
37
|
+
};
|
|
38
|
+
|
|
39
|
+
var minX = minXFromPageItems(page.items);
|
|
40
|
+
page.items.forEach((item) => {
|
|
41
|
+
if (
|
|
42
|
+
stashedBlock.items.length > 0 &&
|
|
43
|
+
shouldFlushBlock(stashedBlock, item, minX, mostUsedDistance)
|
|
44
|
+
) {
|
|
45
|
+
flushStashedItems();
|
|
46
|
+
}
|
|
47
|
+
stashedBlock.addItem(item);
|
|
48
|
+
});
|
|
49
|
+
if (stashedBlock.items.length > 0) {
|
|
50
|
+
flushStashedItems();
|
|
51
|
+
}
|
|
52
|
+
page.items = blocks;
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
return new ParseResult({
|
|
56
|
+
...parseResult,
|
|
57
|
+
messages: [
|
|
58
|
+
"Gathered " +
|
|
59
|
+
createdBlocks +
|
|
60
|
+
" blocks out of " +
|
|
61
|
+
lineItemCount +
|
|
62
|
+
" line items",
|
|
63
|
+
],
|
|
64
|
+
});
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
function shouldFlushBlock(stashedBlock: any, item: any, minX: number, mostUsedDistance: number): boolean {
|
|
69
|
+
if (
|
|
70
|
+
stashedBlock.type &&
|
|
71
|
+
stashedBlock.type.mergeFollowingNonTypedItems &&
|
|
72
|
+
!item.type
|
|
73
|
+
) {
|
|
74
|
+
return false;
|
|
75
|
+
}
|
|
76
|
+
const lastItem = stashedBlock.items[stashedBlock.items.length - 1];
|
|
77
|
+
const hasBigDistance = bigDistance(lastItem, item, minX, mostUsedDistance);
|
|
78
|
+
if (
|
|
79
|
+
stashedBlock.type &&
|
|
80
|
+
stashedBlock.type.mergeFollowingNonTypedItemsWithSmallDistance &&
|
|
81
|
+
!item.type &&
|
|
82
|
+
!hasBigDistance
|
|
83
|
+
) {
|
|
84
|
+
return false;
|
|
85
|
+
}
|
|
86
|
+
if (item.type !== stashedBlock.type) {
|
|
87
|
+
return true;
|
|
88
|
+
}
|
|
89
|
+
if (item.type) {
|
|
90
|
+
return !item.type.mergeToBlock;
|
|
91
|
+
} else {
|
|
92
|
+
return hasBigDistance;
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
function bigDistance(lastItem: any, item: any, minX: number, mostUsedDistance: number): boolean {
|
|
97
|
+
const distance = lastItem.y - item.y;
|
|
98
|
+
if (distance < 0 - mostUsedDistance / 2) {
|
|
99
|
+
// distance is negative - and not only a bit
|
|
100
|
+
return true;
|
|
101
|
+
}
|
|
102
|
+
var allowedDisctance = mostUsedDistance + 1;
|
|
103
|
+
if (lastItem.x > minX && item.x > minX) {
|
|
104
|
+
// intended elements like lists often have greater spacing
|
|
105
|
+
allowedDisctance = mostUsedDistance + mostUsedDistance / 2;
|
|
106
|
+
}
|
|
107
|
+
if (distance > allowedDisctance) {
|
|
108
|
+
return true;
|
|
109
|
+
}
|
|
110
|
+
return false;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
|