lcpcli 0.2.9__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {lcpcli-0.2.9 → lcpcli-0.3.1}/.gitignore +1 -0
  2. {lcpcli-0.2.9 → lcpcli-0.3.1}/LICENSE.txt +1 -1
  3. lcpcli-0.3.1/PKG-INFO +111 -0
  4. lcpcli-0.3.1/README.md +81 -0
  5. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/__init__.py +1 -1
  6. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/builder.py +118 -53
  7. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/check_files.py +42 -14
  8. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/cli.py +37 -8
  9. lcpcli-0.3.1/lcpcli/conllu_builder.py +307 -0
  10. lcpcli-0.3.1/lcpcli/corpert.py +148 -0
  11. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/lcp_corpus_template.json +9 -1
  12. lcpcli-0.3.1/lcpcli/lcp_upload.py +814 -0
  13. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/lcpcli.py +10 -58
  14. lcpcli-0.3.1/lcpcli/utils.py +235 -0
  15. {lcpcli-0.2.9 → lcpcli-0.3.1}/pyproject.toml +3 -3
  16. lcpcli-0.2.9/PKG-INFO +0 -370
  17. lcpcli-0.2.9/README.md +0 -340
  18. lcpcli-0.2.9/lcpcli/corpert.py +0 -617
  19. lcpcli-0.2.9/lcpcli/lcp_upload.py +0 -547
  20. lcpcli-0.2.9/lcpcli/parsers/__init__.py +0 -0
  21. lcpcli-0.2.9/lcpcli/parsers/_parser.py +0 -755
  22. lcpcli-0.2.9/lcpcli/parsers/conllu.py +0 -429
  23. lcpcli-0.2.9/lcpcli/utils.py +0 -769
  24. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/__main__.py +0 -0
  25. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/input/doc.conllu +0 -0
  26. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/input/meta.json +0 -0
  27. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/input/namedentity.csv +0 -0
  28. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/input/shot.csv +0 -0
  29. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/output/media/bunny.mp4 +0 -0
  30. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/input/in.conllu +0 -0
  31. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/input/in.vert +0 -0
  32. {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/input/in_tei_spoken.xml +0 -0
@@ -2,6 +2,7 @@ __pycache__/
2
2
  lcpcli/__pycache__/
3
3
  tmp.py
4
4
  *.pyc
5
+ .env
5
6
  build/
6
7
  dist/
7
8
  dist/
@@ -1,4 +1,4 @@
1
- Copyright 2025 LiRI - UZH
1
+ Copyright 2026 LiRI - UZH
2
2
 
3
3
  Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the “Software”), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
4
4
 
lcpcli-0.3.1/PKG-INFO ADDED
@@ -0,0 +1,111 @@
1
+ Metadata-Version: 2.4
2
+ Name: lcpcli
3
+ Version: 0.3.1
4
+ Summary: Helper for converting CONLLU files and uploading the corpus to LiRI Corpus Platform (LCP)
5
+ Project-URL: Homepage, https://github.com/liri-uzh/lcpcli
6
+ Project-URL: Issues, https://github.com/liri-uzh/lcpcli/issues
7
+ Author-email: Danny McDonald <daniel.mcdonald@uzh.ch>, Igor Mustač <igor.mustac@uzh.ch>, Jeremy Zehr <jeremy.zehr@uzh.ch>, Jonathan Schaber <jeremy.schaber@uzh.ch>
8
+ License: MIT
9
+ License-File: LICENSE.txt
10
+ Keywords: CONLL,TEI,VERT,corpora,corpus,linguistics
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Requires-Python: >=3.10
17
+ Requires-Dist: diskcache>=5.6.3
18
+ Requires-Dist: jsonpickle>=3.0
19
+ Requires-Dist: jsonschema>=4.21
20
+ Requires-Dist: lxml>=4.7.1
21
+ Requires-Dist: pandas>=2.2.2
22
+ Requires-Dist: py7zr>=0.20.5
23
+ Requires-Dist: requests>=2.30.0
24
+ Requires-Dist: tqdm>=4.65.0
25
+ Requires-Dist: tuspy==1.1.0
26
+ Requires-Dist: types-requests>=2.30.0.0
27
+ Requires-Dist: types-tqdm>=4.65.0.1
28
+ Requires-Dist: xmltodict>=0.13
29
+ Description-Content-Type: text/markdown
30
+
31
+ # LCP CLI module
32
+
33
+ > Command-line tool for converting CONLLU files and uploading the corpus to LCP
34
+
35
+ ## Installation
36
+
37
+ Make sure you have python 3.11+ with `pip` installed in your local environment, then run:
38
+
39
+ ```bash
40
+ pip install lcpcli
41
+ ```
42
+
43
+ ## Usage
44
+
45
+ **Examples:**
46
+
47
+ Conversion of a CoNLL-U (Plus) corpus:
48
+
49
+ ```bash
50
+ lcpcli -i ~/conll_ext/ -o ~/upload/
51
+ ```
52
+
53
+ Data upload:
54
+
55
+ ```bash
56
+ lcpcli -c ~/upload/ -k $API_KEY -s $API_SECRET -p "my project" --live
57
+ ```
58
+
59
+ Including `--live` points the upload to the live instance of LCP. Leave it out if you want to add a corpus to an instance of LCP running on `localhost`.
60
+
61
+ **Help:**
62
+
63
+ ```bash
64
+ lcpcli --help
65
+ ```
66
+
67
+ `lcpcli` can take a corpus of CoNLL-U (PLUS) files and import it to a collection created on LCP.
68
+
69
+ Besides the standard token-level CoNLL-U fields (`form`, `lemma`, `upos`, `xpos`, `feats`, `head`, `deprel`, `deps`) one can also provide document-, paragraph- and sentence-level annotations using comment lines in the files (see [the CoNLL-U Format section](#conll-u-format)).
70
+
71
+ ### CoNLL-U Format
72
+
73
+ The CoNLL-U format is documented at: https://universaldependencies.org/format.html
74
+
75
+ The LCP CLI converter will treat all the comments that start with `# newdoc KEY = VALUE` as document-level attributes, and all the comments that start with `# newpar KEY = VALUE` as paragraph-level attributes. All other comment lines following the format `# key = value` will be treated sentence-level attributes.
76
+
77
+ The key-value pairs in the `FEATS` and `MISC` columns of a token line will be mapped to corresponding attributes in the LCP corpus. Additionally, if the `MISC` cell includes `SpaceAfter=Yes` or `SpaceAfter=No` (case senstive) the token will be represented with (respectively, without) a trailing space character in the database.
78
+
79
+ #### CoNLL-U Plus
80
+
81
+ CoNLL-U Plus is an extension to the CoNLLU-U format documented at: https://universaldependencies.org/ext-format.html
82
+
83
+ If your files start with a comment line of the form `# global.columns = ID FORM LEMMA UPOS XPOS FEATS HEAD DEPREL DEPS MISC`, `lcpcli` will treat them as CoNLL-U PLUS files and process the columns according to the names you set in that line.
84
+
85
+ ### CoNLL-U conversion and upload
86
+
87
+ 1. Create a directory in which you have all your properly-fromatted CoNLL-U files.
88
+
89
+ 2. Visit an LCP instance (e.g. _catchphrase_) and create a new collection if you don't already have one where your corpus should go.
90
+
91
+ 3. Retrieve the API key and secret for your project by clicking on the button that says: "Create API Key".
92
+
93
+ 4. Once you have your API key and secret, you can start converting and uploading your corpus by running the following command:
94
+
95
+ ```
96
+ lcpcli -i $CONLLU_FOLDER -o $OUTPUT_FOLDER -k $API_KEY -s $API_SECRET -p $PROJECT_NAME --live
97
+ ```
98
+
99
+ - `$CONLLU_FOLDER` should point to the folder that contains your CONLLU files
100
+ - `$OUTPUT_FOLDER` should point to *another* folder that will be used to store the converted files to be uploaded
101
+ - `$API_KEY` is the key you copied from your project on LCP (still visible when you visit the page)
102
+ - `$API_SECRET` is the secret you copied from your project on LCP (only visible upon API Key creation)
103
+ - `$PROJECT_NAME` is the name of the project exactly as displayed on LCP -- it is case-sensitive, and space characters should be escaped
104
+
105
+ ### Other input formats, rich data
106
+
107
+ Previous versions of `lcpcli` defined procedures to include rich annotations in CoNLL-U files, including time-anchored media files, in combination with annex non-CoNLL-U files. These methods are no longer supported -- use an older version of `lcpcli` if you require those features.
108
+
109
+ `lcpcli` now ships with a Python module called `lcpcli.builder` that you can use to convert any input format. The default CoNLL-U converter included in `lcpcli` uses `lcpcli.builder` under the hood.
110
+
111
+ You can find a short tutorial on how to use the module [in BUILDER.md](BUILDER.md). Further information can be found in [the LCP documentation](https://lcp.linguistik.uzh.ch/manual/builder.html).
lcpcli-0.3.1/README.md ADDED
@@ -0,0 +1,81 @@
1
+ # LCP CLI module
2
+
3
+ > Command-line tool for converting CONLLU files and uploading the corpus to LCP
4
+
5
+ ## Installation
6
+
7
+ Make sure you have python 3.11+ with `pip` installed in your local environment, then run:
8
+
9
+ ```bash
10
+ pip install lcpcli
11
+ ```
12
+
13
+ ## Usage
14
+
15
+ **Examples:**
16
+
17
+ Conversion of a CoNLL-U (Plus) corpus:
18
+
19
+ ```bash
20
+ lcpcli -i ~/conll_ext/ -o ~/upload/
21
+ ```
22
+
23
+ Data upload:
24
+
25
+ ```bash
26
+ lcpcli -c ~/upload/ -k $API_KEY -s $API_SECRET -p "my project" --live
27
+ ```
28
+
29
+ Including `--live` points the upload to the live instance of LCP. Leave it out if you want to add a corpus to an instance of LCP running on `localhost`.
30
+
31
+ **Help:**
32
+
33
+ ```bash
34
+ lcpcli --help
35
+ ```
36
+
37
+ `lcpcli` can take a corpus of CoNLL-U (PLUS) files and import it to a collection created on LCP.
38
+
39
+ Besides the standard token-level CoNLL-U fields (`form`, `lemma`, `upos`, `xpos`, `feats`, `head`, `deprel`, `deps`) one can also provide document-, paragraph- and sentence-level annotations using comment lines in the files (see [the CoNLL-U Format section](#conll-u-format)).
40
+
41
+ ### CoNLL-U Format
42
+
43
+ The CoNLL-U format is documented at: https://universaldependencies.org/format.html
44
+
45
+ The LCP CLI converter will treat all the comments that start with `# newdoc KEY = VALUE` as document-level attributes, and all the comments that start with `# newpar KEY = VALUE` as paragraph-level attributes. All other comment lines following the format `# key = value` will be treated sentence-level attributes.
46
+
47
+ The key-value pairs in the `FEATS` and `MISC` columns of a token line will be mapped to corresponding attributes in the LCP corpus. Additionally, if the `MISC` cell includes `SpaceAfter=Yes` or `SpaceAfter=No` (case senstive) the token will be represented with (respectively, without) a trailing space character in the database.
48
+
49
+ #### CoNLL-U Plus
50
+
51
+ CoNLL-U Plus is an extension to the CoNLLU-U format documented at: https://universaldependencies.org/ext-format.html
52
+
53
+ If your files start with a comment line of the form `# global.columns = ID FORM LEMMA UPOS XPOS FEATS HEAD DEPREL DEPS MISC`, `lcpcli` will treat them as CoNLL-U PLUS files and process the columns according to the names you set in that line.
54
+
55
+ ### CoNLL-U conversion and upload
56
+
57
+ 1. Create a directory in which you have all your properly-fromatted CoNLL-U files.
58
+
59
+ 2. Visit an LCP instance (e.g. _catchphrase_) and create a new collection if you don't already have one where your corpus should go.
60
+
61
+ 3. Retrieve the API key and secret for your project by clicking on the button that says: "Create API Key".
62
+
63
+ 4. Once you have your API key and secret, you can start converting and uploading your corpus by running the following command:
64
+
65
+ ```
66
+ lcpcli -i $CONLLU_FOLDER -o $OUTPUT_FOLDER -k $API_KEY -s $API_SECRET -p $PROJECT_NAME --live
67
+ ```
68
+
69
+ - `$CONLLU_FOLDER` should point to the folder that contains your CONLLU files
70
+ - `$OUTPUT_FOLDER` should point to *another* folder that will be used to store the converted files to be uploaded
71
+ - `$API_KEY` is the key you copied from your project on LCP (still visible when you visit the page)
72
+ - `$API_SECRET` is the secret you copied from your project on LCP (only visible upon API Key creation)
73
+ - `$PROJECT_NAME` is the name of the project exactly as displayed on LCP -- it is case-sensitive, and space characters should be escaped
74
+
75
+ ### Other input formats, rich data
76
+
77
+ Previous versions of `lcpcli` defined procedures to include rich annotations in CoNLL-U files, including time-anchored media files, in combination with annex non-CoNLL-U files. These methods are no longer supported -- use an older version of `lcpcli` if you require those features.
78
+
79
+ `lcpcli` now ships with a Python module called `lcpcli.builder` that you can use to convert any input format. The default CoNLL-U converter included in `lcpcli` uses `lcpcli.builder` under the hood.
80
+
81
+ You can find a short tutorial on how to use the module [in BUILDER.md](BUILDER.md). Further information can be found in [the LCP documentation](https://lcp.linguistik.uzh.ch/manual/builder.html).
@@ -1,3 +1,3 @@
1
- __version__ = "0.2.9"
1
+ __version__ = "0.3.1"
2
2
 
3
3
  from .lcpcli import Lcpcli # noqa: F401
@@ -4,8 +4,8 @@ import csv
4
4
  import json
5
5
  import os
6
6
  import re
7
- import tempfile
8
7
  import shutil
8
+ import tempfile
9
9
 
10
10
  from typing import Any
11
11
  from uuid import uuid4
@@ -16,9 +16,9 @@ ANCHORINGS = ("stream", "time", "location")
16
16
  # ATYPES = ("text", "categorical", "number", "dict", "labels")
17
17
  ATYPES_LOOKUP = ("text", "dict", "labels")
18
18
  NAMEDATALEN = 63
19
- PATTERN_TXT = (
20
- "(must start with a lower case and only contain alpha-numerical characters)"
21
- )
19
+ PATTERN_TXT = "(must start with a lower case, be at leat 2 characters long and only contain alpha-numerical characters)"
20
+
21
+ IS_NUM = re.compile(r"^[0-9]+(\.[0-9]+)?$")
22
22
 
23
23
 
24
24
  def meta_subattr(meta: dict, k: str, v: Any) -> dict:
@@ -28,11 +28,7 @@ def meta_subattr(meta: dict, k: str, v: Any) -> dict:
28
28
  sub_attr = meta.setdefault(k, {})
29
29
  if isinstance(v, list):
30
30
  sub_attr["type"] = "labels"
31
- elif (
32
- isinstance(v, (int, float))
33
- or isinstance(v, str)
34
- and v.replace(".", "", 1).isdigit()
35
- ):
31
+ elif isinstance(v, (int, float)) or isinstance(v, str) and IS_NUM.match(v):
36
32
  sub_attr["type"] = "text" if sub_attr.get("type") == "text" else "number"
37
33
  elif isinstance(v, dict):
38
34
  sub_attr["type"] = "dict"
@@ -50,14 +46,17 @@ def get_layer_method(layer: "Layer"):
50
46
  corpus._layers.pop(layer._name, "")
51
47
  fname = f"{layer._name.lower()}.csv"
52
48
  corpus._files[fname].close()
49
+ os.unlink(corpus._files[fname].name)
53
50
  corpus._files.pop(fname)
54
51
  return GlobalAttribute(corpus, layer._name, args[0])
55
52
  largs = [a for a in args]
56
- if layer._name == corpus._token and isinstance(largs[0], str):
53
+ if layer._name == corpus._token and largs and isinstance(largs[0], str):
57
54
  form = largs.pop(0)
58
55
  layer.form = form
59
56
  if len(largs) > 0:
60
- assert all(isinstance(c, Layer) for c in largs), RuntimeError()
57
+ assert all(isinstance(c, Layer) for c in largs), RuntimeError(
58
+ "Tried to pass non-layers as arguments of a layer"
59
+ )
61
60
  layer.add(*largs)
62
61
  for aname, avalue in kwargs.items():
63
62
  setattr(layer, aname, avalue)
@@ -85,10 +84,23 @@ def get_layer_method(layer: "Layer"):
85
84
  a.make()
86
85
  return
87
86
  # source is the attribute that's missing in at least one layer
88
- source_a = next(
89
- ra for ra in relation_attrs if any(ra not in a._attributes for a in args)
90
- )
91
- target_a = next(ra for ra in relation_attrs if ra != source_a)
87
+ try:
88
+ source_a = next(
89
+ ra
90
+ for ra in relation_attrs
91
+ if any(ra not in a._attributes for a in args)
92
+ )
93
+ except:
94
+ # if can't identify a missing attribute,
95
+ # look for "head"/"source" or use the first one
96
+ source_a = next(
97
+ (ra for ra in relation_attrs if ra in ("head", "source")),
98
+ next(ra for ra in relation_attrs),
99
+ )
100
+ target_a = next((ra for ra in relation_attrs if ra != source_a), None)
101
+ if not target_a:
102
+ # no target: possibly a one-token sentence, return
103
+ return
92
104
  # reference nested sets by target's id
93
105
  nested_sets = {
94
106
  a._attributes[target_a]._value._id: NestedSet(
@@ -246,7 +258,7 @@ class Corpus:
246
258
  can_categorize = (
247
259
  not (is_token and aname in ("form", "lemma"))
248
260
  and len(lookup) <= 100
249
- and all(len(v) < NAMEDATALEN for v in lookup)
261
+ and all(len(str(v)) < NAMEDATALEN for v in lookup)
250
262
  )
251
263
  if can_categorize:
252
264
  texts_to_categorical[na] = aname
@@ -326,7 +338,7 @@ class Corpus:
326
338
  "corpusDescription": self._corpus_description,
327
339
  "date": self._date,
328
340
  "url": self._url,
329
- "revision": 1,
341
+ "revision": self._revision,
330
342
  },
331
343
  "firstClass": {
332
344
  "token": self._token,
@@ -374,7 +386,11 @@ class Corpus:
374
386
  if ais_global:
375
387
  aopts["isGlobal"] = True
376
388
  if aopts["type"] == "categorical" and not ais_global:
377
- aopts["values"] = [v for v in mapping.lookups[aname] if v]
389
+ aopts["values"] = [
390
+ str(v)
391
+ for v in mapping.lookups[aname]
392
+ if v is not None and v != ""
393
+ ]
378
394
  elif aopts["type"] == "ref":
379
395
  aopts.pop("type")
380
396
  aopts.pop("nullable", "")
@@ -418,14 +434,23 @@ class Layer:
418
434
  assert re.match(r"[a-z][a-zA-Z0-9_]+$", name), RuntimeError(
419
435
  f"The attribute '{name}' on the layer {self._name} does not match the pattern {PATTERN_TXT}"
420
436
  )
437
+ # Disallow linebreak in token string values because it messes with CSV's (in particular, FTS)
438
+ if (
439
+ self._name == self._corpus._token
440
+ and isinstance(value, str)
441
+ and ("\n" in value or "\r" in value)
442
+ ):
443
+ print(
444
+ f"Warning: a token attribute contains a linebreak; this is not allowed, removing the linebreaks from the value {value}."
445
+ )
446
+ value = value.replace("\n", "").replace("\r", "")
421
447
  Attribute(self, name, value)
422
448
 
423
449
  def __getattribute__(self, name: str):
424
450
  if re.match(r"[A-Z]", name):
425
451
  corpus = self._corpus
426
452
  layer = corpus._add_layer(name)
427
- self._contains.append(layer)
428
- layer._parents.append(self)
453
+ self.add(layer)
429
454
  return get_layer_method(layer)
430
455
  return super().__getattribute__(name)
431
456
 
@@ -471,7 +496,47 @@ class Layer:
471
496
  ch.append(c)
472
497
  return ch
473
498
 
474
- def make(self):
499
+ def _update_parents_anchors(self):
500
+ """Update the anchors of all the parents (recursively)"""
501
+ if not self._made:
502
+ return
503
+ corpus = self._corpus
504
+ parents = self._parents
505
+ while parents:
506
+ current_parents = [*parents]
507
+ parents = []
508
+ for parent in current_parents:
509
+ parents += parent._parents
510
+ for anc_name, anchors in self._anchorings.items():
511
+ if anc_name not in parent._anchorings:
512
+ parent._anchorings[anc_name] = [*anchors]
513
+ parent_anchors = parent._anchorings[anc_name]
514
+ if anchors[0] < parent_anchors[0]:
515
+ parent_anchors[0] = anchors[0]
516
+ if anc_name == "time" and parent._name == corpus._document:
517
+ if corpus._upperFrameDocument < parent_anchors[0]:
518
+ parent_anchors[0] = corpus._upperFrameDocument
519
+ corpus._upperFrameDocument = parent_anchors[1]
520
+ if anc_name != "location":
521
+ if anchors[1] > parent_anchors[1]:
522
+ parent_anchors[1] = anchors[1]
523
+ continue
524
+ if anchors[1] < parent_anchors[1]:
525
+ parent_anchors[1] = anchors[1]
526
+ if anchors[2] > parent_anchors[2]:
527
+ parent_anchors[2] = anchors[2]
528
+ if anchors[3] > parent_anchors[3]:
529
+ parent_anchors[3] = anchors[3]
530
+
531
+ def clear(self):
532
+ if not self._made:
533
+ return
534
+ # Prepare for deletion: no pointers to other layers/global attributes
535
+ self._parents = []
536
+ self._contains = []
537
+ self._attributes = {}
538
+
539
+ def make(self, clear=False):
475
540
  if self._made:
476
541
  return
477
542
  corpus = self._corpus
@@ -494,6 +559,9 @@ class Layer:
494
559
  rows.append(doc_name)
495
560
  rows.append(json.dumps(self._media))
496
561
  if is_token:
562
+ assert "form" in self._attributes, RuntimeError(
563
+ "Tried to make a token with no form"
564
+ )
497
565
  seg_parent = self._find_in_parents(corpus._segment)
498
566
  rows.append(seg_parent._id)
499
567
  char_low = corpus._char_counter
@@ -502,35 +570,8 @@ class Layer:
502
570
  )
503
571
  self._anchorings["stream"] = [char_low, corpus._char_counter]
504
572
  elif self._contains:
505
- unset_anchorings = {a for a in ANCHORINGS if not self._anchorings.get(a)}
506
573
  for child in self._contains:
507
574
  child.make()
508
- if child._name not in mapping.contains:
509
- mapping.contains.append(child._name)
510
- # Anchorings
511
- for a in unset_anchorings:
512
- if a not in child._anchorings:
513
- continue
514
- child_a = child._anchorings[a]
515
- if a not in self._anchorings:
516
- self._anchorings[a] = [*child_a]
517
- self_a = self._anchorings[a]
518
- if child_a[0] < self_a[0]:
519
- self_a[0] = child_a[0]
520
- if a == "time" and self._name == corpus._document:
521
- if corpus._upperFrameDocument < self_a[0]:
522
- self_a[0] = corpus._upperFrameDocument
523
- corpus._upperFrameDocument = self_a[1]
524
- if a != "location":
525
- if child_a[1] > self_a[1]:
526
- self_a[1] = child_a[1]
527
- continue
528
- if child_a[1] < self_a[1]:
529
- self_a[1] = child_a[1]
530
- if child_a[2] > self_a[2]:
531
- self_a[2] = child_a[2]
532
- if child_a[3] > self_a[3]:
533
- self_a[3] = child_a[3]
534
575
  if is_segment:
535
576
  tokens = [
536
577
  ch._attributes.values()
@@ -568,15 +609,27 @@ class Layer:
568
609
  rows.append(v)
569
610
  # Add any new attribute to mapping
570
611
  for aname, attr in self._attributes.items():
612
+ atype = attr._type
571
613
  if aname in mapping.attributes:
614
+ mattr = mapping.attributes[aname]
615
+ if atype == "text" and mattr["type"] != atype:
616
+ mattr["type"] = "text"
617
+ else:
618
+ try:
619
+ mapping.attributes[aname]["subtype"] = attr._subtype
620
+ except:
621
+ pass
572
622
  continue
573
- atype = attr._type
574
623
  mapping.attributes[aname] = {
575
624
  "type": atype,
576
625
  "nullable": (
577
626
  True if mapping.counter > 1 else False
578
627
  ), # adding a new attribute
579
628
  }
629
+ try:
630
+ mapping.attributes[aname]["subtype"] = attr._subtype
631
+ except:
632
+ pass
580
633
  if atype == "ref":
581
634
  mapping.attributes[aname]["ref"] = attr._ref.lower()
582
635
  elif atype in ATYPES_LOOKUP and aname != "meta":
@@ -656,6 +709,9 @@ class Layer:
656
709
  rows.append("" if val == None else str(val))
657
710
  mapping.csvs["_main"].writerow(rows)
658
711
  self._made = True
712
+ self._update_parents_anchors()
713
+ if clear:
714
+ self.clear()
659
715
  return self
660
716
 
661
717
  def set_time(self, *args):
@@ -711,19 +767,25 @@ class Layer:
711
767
  return self
712
768
 
713
769
  def add(self, *layers: "Layer"):
714
- assert not self._contains or all(
715
- l._name == self._contains[0]._name for l in layers
716
- ), RuntimeError("All the children of a layer must be of the same type")
770
+ # assert not self._contains or all(
771
+ # l._name == self._contains[0]._name for l in layers
772
+ # ), RuntimeError("All the children of a layer must be of the same type")
717
773
  self._contains += layers
774
+ mapping = self._corpus._layers[self._name]
718
775
  for layer in layers:
776
+ if layer._name not in mapping.contains:
777
+ mapping.contains.append(layer._name)
719
778
  if self not in layer._parents:
720
779
  layer._parents.append(self)
780
+ layer._update_parents_anchors()
721
781
  return self
722
782
 
723
783
 
724
784
  class Attribute:
725
785
  def __init__(self, layer: Layer, name: str, value: Any = None):
726
786
  self._name = name
787
+ if name not in layer._attributes:
788
+ layer._attributes[name] = self
727
789
  self._value = value
728
790
  self._layer = layer
729
791
  self._ref = None
@@ -739,6 +801,10 @@ class Attribute:
739
801
  self._value = json.dumps(sorted_dict(value))
740
802
  elif isinstance(value, (int, float)):
741
803
  atype = "number"
804
+ if isinstance(value, float):
805
+ self._subtype = "float"
806
+ # overwrite to ensure subtype is taken into consideration
807
+ layer._attributes[name] = self
742
808
  elif isinstance(value, GlobalAttribute):
743
809
  atype = "ref"
744
810
  self._ref = value._name
@@ -747,7 +813,6 @@ class Attribute:
747
813
  atype = "entity"
748
814
  self._ref = value._name
749
815
  self._type: str = atype
750
- layer._attributes[name] = self
751
816
 
752
817
 
753
818
  class GlobalAttribute: