lcpcli 0.2.9__tar.gz → 0.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {lcpcli-0.2.9 → lcpcli-0.3.1}/.gitignore +1 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/LICENSE.txt +1 -1
- lcpcli-0.3.1/PKG-INFO +111 -0
- lcpcli-0.3.1/README.md +81 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/__init__.py +1 -1
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/builder.py +118 -53
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/check_files.py +42 -14
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/cli.py +37 -8
- lcpcli-0.3.1/lcpcli/conllu_builder.py +307 -0
- lcpcli-0.3.1/lcpcli/corpert.py +148 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/lcp_corpus_template.json +9 -1
- lcpcli-0.3.1/lcpcli/lcp_upload.py +814 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/lcpcli.py +10 -58
- lcpcli-0.3.1/lcpcli/utils.py +235 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/pyproject.toml +3 -3
- lcpcli-0.2.9/PKG-INFO +0 -370
- lcpcli-0.2.9/README.md +0 -340
- lcpcli-0.2.9/lcpcli/corpert.py +0 -617
- lcpcli-0.2.9/lcpcli/lcp_upload.py +0 -547
- lcpcli-0.2.9/lcpcli/parsers/__init__.py +0 -0
- lcpcli-0.2.9/lcpcli/parsers/_parser.py +0 -755
- lcpcli-0.2.9/lcpcli/parsers/conllu.py +0 -429
- lcpcli-0.2.9/lcpcli/utils.py +0 -769
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/__main__.py +0 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/input/doc.conllu +0 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/input/meta.json +0 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/input/namedentity.csv +0 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/input/shot.csv +0 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/free_video_corpus/output/media/bunny.mp4 +0 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/input/in.conllu +0 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/input/in.vert +0 -0
- {lcpcli-0.2.9 → lcpcli-0.3.1}/lcpcli/data/input/in_tei_spoken.xml +0 -0
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
Copyright
|
|
1
|
+
Copyright 2026 LiRI - UZH
|
|
2
2
|
|
|
3
3
|
Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the “Software”), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
|
|
4
4
|
|
lcpcli-0.3.1/PKG-INFO
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: lcpcli
|
|
3
|
+
Version: 0.3.1
|
|
4
|
+
Summary: Helper for converting CONLLU files and uploading the corpus to LiRI Corpus Platform (LCP)
|
|
5
|
+
Project-URL: Homepage, https://github.com/liri-uzh/lcpcli
|
|
6
|
+
Project-URL: Issues, https://github.com/liri-uzh/lcpcli/issues
|
|
7
|
+
Author-email: Danny McDonald <daniel.mcdonald@uzh.ch>, Igor Mustač <igor.mustac@uzh.ch>, Jeremy Zehr <jeremy.zehr@uzh.ch>, Jonathan Schaber <jeremy.schaber@uzh.ch>
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE.txt
|
|
10
|
+
Keywords: CONLL,TEI,VERT,corpora,corpus,linguistics
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Requires-Dist: diskcache>=5.6.3
|
|
18
|
+
Requires-Dist: jsonpickle>=3.0
|
|
19
|
+
Requires-Dist: jsonschema>=4.21
|
|
20
|
+
Requires-Dist: lxml>=4.7.1
|
|
21
|
+
Requires-Dist: pandas>=2.2.2
|
|
22
|
+
Requires-Dist: py7zr>=0.20.5
|
|
23
|
+
Requires-Dist: requests>=2.30.0
|
|
24
|
+
Requires-Dist: tqdm>=4.65.0
|
|
25
|
+
Requires-Dist: tuspy==1.1.0
|
|
26
|
+
Requires-Dist: types-requests>=2.30.0.0
|
|
27
|
+
Requires-Dist: types-tqdm>=4.65.0.1
|
|
28
|
+
Requires-Dist: xmltodict>=0.13
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# LCP CLI module
|
|
32
|
+
|
|
33
|
+
> Command-line tool for converting CONLLU files and uploading the corpus to LCP
|
|
34
|
+
|
|
35
|
+
## Installation
|
|
36
|
+
|
|
37
|
+
Make sure you have python 3.11+ with `pip` installed in your local environment, then run:
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
pip install lcpcli
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Usage
|
|
44
|
+
|
|
45
|
+
**Examples:**
|
|
46
|
+
|
|
47
|
+
Conversion of a CoNLL-U (Plus) corpus:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
lcpcli -i ~/conll_ext/ -o ~/upload/
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Data upload:
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
lcpcli -c ~/upload/ -k $API_KEY -s $API_SECRET -p "my project" --live
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Including `--live` points the upload to the live instance of LCP. Leave it out if you want to add a corpus to an instance of LCP running on `localhost`.
|
|
60
|
+
|
|
61
|
+
**Help:**
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
lcpcli --help
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
`lcpcli` can take a corpus of CoNLL-U (PLUS) files and import it to a collection created on LCP.
|
|
68
|
+
|
|
69
|
+
Besides the standard token-level CoNLL-U fields (`form`, `lemma`, `upos`, `xpos`, `feats`, `head`, `deprel`, `deps`) one can also provide document-, paragraph- and sentence-level annotations using comment lines in the files (see [the CoNLL-U Format section](#conll-u-format)).
|
|
70
|
+
|
|
71
|
+
### CoNLL-U Format
|
|
72
|
+
|
|
73
|
+
The CoNLL-U format is documented at: https://universaldependencies.org/format.html
|
|
74
|
+
|
|
75
|
+
The LCP CLI converter will treat all the comments that start with `# newdoc KEY = VALUE` as document-level attributes, and all the comments that start with `# newpar KEY = VALUE` as paragraph-level attributes. All other comment lines following the format `# key = value` will be treated sentence-level attributes.
|
|
76
|
+
|
|
77
|
+
The key-value pairs in the `FEATS` and `MISC` columns of a token line will be mapped to corresponding attributes in the LCP corpus. Additionally, if the `MISC` cell includes `SpaceAfter=Yes` or `SpaceAfter=No` (case senstive) the token will be represented with (respectively, without) a trailing space character in the database.
|
|
78
|
+
|
|
79
|
+
#### CoNLL-U Plus
|
|
80
|
+
|
|
81
|
+
CoNLL-U Plus is an extension to the CoNLLU-U format documented at: https://universaldependencies.org/ext-format.html
|
|
82
|
+
|
|
83
|
+
If your files start with a comment line of the form `# global.columns = ID FORM LEMMA UPOS XPOS FEATS HEAD DEPREL DEPS MISC`, `lcpcli` will treat them as CoNLL-U PLUS files and process the columns according to the names you set in that line.
|
|
84
|
+
|
|
85
|
+
### CoNLL-U conversion and upload
|
|
86
|
+
|
|
87
|
+
1. Create a directory in which you have all your properly-fromatted CoNLL-U files.
|
|
88
|
+
|
|
89
|
+
2. Visit an LCP instance (e.g. _catchphrase_) and create a new collection if you don't already have one where your corpus should go.
|
|
90
|
+
|
|
91
|
+
3. Retrieve the API key and secret for your project by clicking on the button that says: "Create API Key".
|
|
92
|
+
|
|
93
|
+
4. Once you have your API key and secret, you can start converting and uploading your corpus by running the following command:
|
|
94
|
+
|
|
95
|
+
```
|
|
96
|
+
lcpcli -i $CONLLU_FOLDER -o $OUTPUT_FOLDER -k $API_KEY -s $API_SECRET -p $PROJECT_NAME --live
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
- `$CONLLU_FOLDER` should point to the folder that contains your CONLLU files
|
|
100
|
+
- `$OUTPUT_FOLDER` should point to *another* folder that will be used to store the converted files to be uploaded
|
|
101
|
+
- `$API_KEY` is the key you copied from your project on LCP (still visible when you visit the page)
|
|
102
|
+
- `$API_SECRET` is the secret you copied from your project on LCP (only visible upon API Key creation)
|
|
103
|
+
- `$PROJECT_NAME` is the name of the project exactly as displayed on LCP -- it is case-sensitive, and space characters should be escaped
|
|
104
|
+
|
|
105
|
+
### Other input formats, rich data
|
|
106
|
+
|
|
107
|
+
Previous versions of `lcpcli` defined procedures to include rich annotations in CoNLL-U files, including time-anchored media files, in combination with annex non-CoNLL-U files. These methods are no longer supported -- use an older version of `lcpcli` if you require those features.
|
|
108
|
+
|
|
109
|
+
`lcpcli` now ships with a Python module called `lcpcli.builder` that you can use to convert any input format. The default CoNLL-U converter included in `lcpcli` uses `lcpcli.builder` under the hood.
|
|
110
|
+
|
|
111
|
+
You can find a short tutorial on how to use the module [in BUILDER.md](BUILDER.md). Further information can be found in [the LCP documentation](https://lcp.linguistik.uzh.ch/manual/builder.html).
|
lcpcli-0.3.1/README.md
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# LCP CLI module
|
|
2
|
+
|
|
3
|
+
> Command-line tool for converting CONLLU files and uploading the corpus to LCP
|
|
4
|
+
|
|
5
|
+
## Installation
|
|
6
|
+
|
|
7
|
+
Make sure you have python 3.11+ with `pip` installed in your local environment, then run:
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install lcpcli
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Usage
|
|
14
|
+
|
|
15
|
+
**Examples:**
|
|
16
|
+
|
|
17
|
+
Conversion of a CoNLL-U (Plus) corpus:
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
lcpcli -i ~/conll_ext/ -o ~/upload/
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Data upload:
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
lcpcli -c ~/upload/ -k $API_KEY -s $API_SECRET -p "my project" --live
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Including `--live` points the upload to the live instance of LCP. Leave it out if you want to add a corpus to an instance of LCP running on `localhost`.
|
|
30
|
+
|
|
31
|
+
**Help:**
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
lcpcli --help
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
`lcpcli` can take a corpus of CoNLL-U (PLUS) files and import it to a collection created on LCP.
|
|
38
|
+
|
|
39
|
+
Besides the standard token-level CoNLL-U fields (`form`, `lemma`, `upos`, `xpos`, `feats`, `head`, `deprel`, `deps`) one can also provide document-, paragraph- and sentence-level annotations using comment lines in the files (see [the CoNLL-U Format section](#conll-u-format)).
|
|
40
|
+
|
|
41
|
+
### CoNLL-U Format
|
|
42
|
+
|
|
43
|
+
The CoNLL-U format is documented at: https://universaldependencies.org/format.html
|
|
44
|
+
|
|
45
|
+
The LCP CLI converter will treat all the comments that start with `# newdoc KEY = VALUE` as document-level attributes, and all the comments that start with `# newpar KEY = VALUE` as paragraph-level attributes. All other comment lines following the format `# key = value` will be treated sentence-level attributes.
|
|
46
|
+
|
|
47
|
+
The key-value pairs in the `FEATS` and `MISC` columns of a token line will be mapped to corresponding attributes in the LCP corpus. Additionally, if the `MISC` cell includes `SpaceAfter=Yes` or `SpaceAfter=No` (case senstive) the token will be represented with (respectively, without) a trailing space character in the database.
|
|
48
|
+
|
|
49
|
+
#### CoNLL-U Plus
|
|
50
|
+
|
|
51
|
+
CoNLL-U Plus is an extension to the CoNLLU-U format documented at: https://universaldependencies.org/ext-format.html
|
|
52
|
+
|
|
53
|
+
If your files start with a comment line of the form `# global.columns = ID FORM LEMMA UPOS XPOS FEATS HEAD DEPREL DEPS MISC`, `lcpcli` will treat them as CoNLL-U PLUS files and process the columns according to the names you set in that line.
|
|
54
|
+
|
|
55
|
+
### CoNLL-U conversion and upload
|
|
56
|
+
|
|
57
|
+
1. Create a directory in which you have all your properly-fromatted CoNLL-U files.
|
|
58
|
+
|
|
59
|
+
2. Visit an LCP instance (e.g. _catchphrase_) and create a new collection if you don't already have one where your corpus should go.
|
|
60
|
+
|
|
61
|
+
3. Retrieve the API key and secret for your project by clicking on the button that says: "Create API Key".
|
|
62
|
+
|
|
63
|
+
4. Once you have your API key and secret, you can start converting and uploading your corpus by running the following command:
|
|
64
|
+
|
|
65
|
+
```
|
|
66
|
+
lcpcli -i $CONLLU_FOLDER -o $OUTPUT_FOLDER -k $API_KEY -s $API_SECRET -p $PROJECT_NAME --live
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
- `$CONLLU_FOLDER` should point to the folder that contains your CONLLU files
|
|
70
|
+
- `$OUTPUT_FOLDER` should point to *another* folder that will be used to store the converted files to be uploaded
|
|
71
|
+
- `$API_KEY` is the key you copied from your project on LCP (still visible when you visit the page)
|
|
72
|
+
- `$API_SECRET` is the secret you copied from your project on LCP (only visible upon API Key creation)
|
|
73
|
+
- `$PROJECT_NAME` is the name of the project exactly as displayed on LCP -- it is case-sensitive, and space characters should be escaped
|
|
74
|
+
|
|
75
|
+
### Other input formats, rich data
|
|
76
|
+
|
|
77
|
+
Previous versions of `lcpcli` defined procedures to include rich annotations in CoNLL-U files, including time-anchored media files, in combination with annex non-CoNLL-U files. These methods are no longer supported -- use an older version of `lcpcli` if you require those features.
|
|
78
|
+
|
|
79
|
+
`lcpcli` now ships with a Python module called `lcpcli.builder` that you can use to convert any input format. The default CoNLL-U converter included in `lcpcli` uses `lcpcli.builder` under the hood.
|
|
80
|
+
|
|
81
|
+
You can find a short tutorial on how to use the module [in BUILDER.md](BUILDER.md). Further information can be found in [the LCP documentation](https://lcp.linguistik.uzh.ch/manual/builder.html).
|
|
@@ -4,8 +4,8 @@ import csv
|
|
|
4
4
|
import json
|
|
5
5
|
import os
|
|
6
6
|
import re
|
|
7
|
-
import tempfile
|
|
8
7
|
import shutil
|
|
8
|
+
import tempfile
|
|
9
9
|
|
|
10
10
|
from typing import Any
|
|
11
11
|
from uuid import uuid4
|
|
@@ -16,9 +16,9 @@ ANCHORINGS = ("stream", "time", "location")
|
|
|
16
16
|
# ATYPES = ("text", "categorical", "number", "dict", "labels")
|
|
17
17
|
ATYPES_LOOKUP = ("text", "dict", "labels")
|
|
18
18
|
NAMEDATALEN = 63
|
|
19
|
-
PATTERN_TXT = (
|
|
20
|
-
|
|
21
|
-
)
|
|
19
|
+
PATTERN_TXT = "(must start with a lower case, be at leat 2 characters long and only contain alpha-numerical characters)"
|
|
20
|
+
|
|
21
|
+
IS_NUM = re.compile(r"^[0-9]+(\.[0-9]+)?$")
|
|
22
22
|
|
|
23
23
|
|
|
24
24
|
def meta_subattr(meta: dict, k: str, v: Any) -> dict:
|
|
@@ -28,11 +28,7 @@ def meta_subattr(meta: dict, k: str, v: Any) -> dict:
|
|
|
28
28
|
sub_attr = meta.setdefault(k, {})
|
|
29
29
|
if isinstance(v, list):
|
|
30
30
|
sub_attr["type"] = "labels"
|
|
31
|
-
elif (
|
|
32
|
-
isinstance(v, (int, float))
|
|
33
|
-
or isinstance(v, str)
|
|
34
|
-
and v.replace(".", "", 1).isdigit()
|
|
35
|
-
):
|
|
31
|
+
elif isinstance(v, (int, float)) or isinstance(v, str) and IS_NUM.match(v):
|
|
36
32
|
sub_attr["type"] = "text" if sub_attr.get("type") == "text" else "number"
|
|
37
33
|
elif isinstance(v, dict):
|
|
38
34
|
sub_attr["type"] = "dict"
|
|
@@ -50,14 +46,17 @@ def get_layer_method(layer: "Layer"):
|
|
|
50
46
|
corpus._layers.pop(layer._name, "")
|
|
51
47
|
fname = f"{layer._name.lower()}.csv"
|
|
52
48
|
corpus._files[fname].close()
|
|
49
|
+
os.unlink(corpus._files[fname].name)
|
|
53
50
|
corpus._files.pop(fname)
|
|
54
51
|
return GlobalAttribute(corpus, layer._name, args[0])
|
|
55
52
|
largs = [a for a in args]
|
|
56
|
-
if layer._name == corpus._token and isinstance(largs[0], str):
|
|
53
|
+
if layer._name == corpus._token and largs and isinstance(largs[0], str):
|
|
57
54
|
form = largs.pop(0)
|
|
58
55
|
layer.form = form
|
|
59
56
|
if len(largs) > 0:
|
|
60
|
-
assert all(isinstance(c, Layer) for c in largs), RuntimeError(
|
|
57
|
+
assert all(isinstance(c, Layer) for c in largs), RuntimeError(
|
|
58
|
+
"Tried to pass non-layers as arguments of a layer"
|
|
59
|
+
)
|
|
61
60
|
layer.add(*largs)
|
|
62
61
|
for aname, avalue in kwargs.items():
|
|
63
62
|
setattr(layer, aname, avalue)
|
|
@@ -85,10 +84,23 @@ def get_layer_method(layer: "Layer"):
|
|
|
85
84
|
a.make()
|
|
86
85
|
return
|
|
87
86
|
# source is the attribute that's missing in at least one layer
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
87
|
+
try:
|
|
88
|
+
source_a = next(
|
|
89
|
+
ra
|
|
90
|
+
for ra in relation_attrs
|
|
91
|
+
if any(ra not in a._attributes for a in args)
|
|
92
|
+
)
|
|
93
|
+
except:
|
|
94
|
+
# if can't identify a missing attribute,
|
|
95
|
+
# look for "head"/"source" or use the first one
|
|
96
|
+
source_a = next(
|
|
97
|
+
(ra for ra in relation_attrs if ra in ("head", "source")),
|
|
98
|
+
next(ra for ra in relation_attrs),
|
|
99
|
+
)
|
|
100
|
+
target_a = next((ra for ra in relation_attrs if ra != source_a), None)
|
|
101
|
+
if not target_a:
|
|
102
|
+
# no target: possibly a one-token sentence, return
|
|
103
|
+
return
|
|
92
104
|
# reference nested sets by target's id
|
|
93
105
|
nested_sets = {
|
|
94
106
|
a._attributes[target_a]._value._id: NestedSet(
|
|
@@ -246,7 +258,7 @@ class Corpus:
|
|
|
246
258
|
can_categorize = (
|
|
247
259
|
not (is_token and aname in ("form", "lemma"))
|
|
248
260
|
and len(lookup) <= 100
|
|
249
|
-
and all(len(v) < NAMEDATALEN for v in lookup)
|
|
261
|
+
and all(len(str(v)) < NAMEDATALEN for v in lookup)
|
|
250
262
|
)
|
|
251
263
|
if can_categorize:
|
|
252
264
|
texts_to_categorical[na] = aname
|
|
@@ -326,7 +338,7 @@ class Corpus:
|
|
|
326
338
|
"corpusDescription": self._corpus_description,
|
|
327
339
|
"date": self._date,
|
|
328
340
|
"url": self._url,
|
|
329
|
-
"revision":
|
|
341
|
+
"revision": self._revision,
|
|
330
342
|
},
|
|
331
343
|
"firstClass": {
|
|
332
344
|
"token": self._token,
|
|
@@ -374,7 +386,11 @@ class Corpus:
|
|
|
374
386
|
if ais_global:
|
|
375
387
|
aopts["isGlobal"] = True
|
|
376
388
|
if aopts["type"] == "categorical" and not ais_global:
|
|
377
|
-
aopts["values"] = [
|
|
389
|
+
aopts["values"] = [
|
|
390
|
+
str(v)
|
|
391
|
+
for v in mapping.lookups[aname]
|
|
392
|
+
if v is not None and v != ""
|
|
393
|
+
]
|
|
378
394
|
elif aopts["type"] == "ref":
|
|
379
395
|
aopts.pop("type")
|
|
380
396
|
aopts.pop("nullable", "")
|
|
@@ -418,14 +434,23 @@ class Layer:
|
|
|
418
434
|
assert re.match(r"[a-z][a-zA-Z0-9_]+$", name), RuntimeError(
|
|
419
435
|
f"The attribute '{name}' on the layer {self._name} does not match the pattern {PATTERN_TXT}"
|
|
420
436
|
)
|
|
437
|
+
# Disallow linebreak in token string values because it messes with CSV's (in particular, FTS)
|
|
438
|
+
if (
|
|
439
|
+
self._name == self._corpus._token
|
|
440
|
+
and isinstance(value, str)
|
|
441
|
+
and ("\n" in value or "\r" in value)
|
|
442
|
+
):
|
|
443
|
+
print(
|
|
444
|
+
f"Warning: a token attribute contains a linebreak; this is not allowed, removing the linebreaks from the value {value}."
|
|
445
|
+
)
|
|
446
|
+
value = value.replace("\n", "").replace("\r", "")
|
|
421
447
|
Attribute(self, name, value)
|
|
422
448
|
|
|
423
449
|
def __getattribute__(self, name: str):
|
|
424
450
|
if re.match(r"[A-Z]", name):
|
|
425
451
|
corpus = self._corpus
|
|
426
452
|
layer = corpus._add_layer(name)
|
|
427
|
-
self.
|
|
428
|
-
layer._parents.append(self)
|
|
453
|
+
self.add(layer)
|
|
429
454
|
return get_layer_method(layer)
|
|
430
455
|
return super().__getattribute__(name)
|
|
431
456
|
|
|
@@ -471,7 +496,47 @@ class Layer:
|
|
|
471
496
|
ch.append(c)
|
|
472
497
|
return ch
|
|
473
498
|
|
|
474
|
-
def
|
|
499
|
+
def _update_parents_anchors(self):
|
|
500
|
+
"""Update the anchors of all the parents (recursively)"""
|
|
501
|
+
if not self._made:
|
|
502
|
+
return
|
|
503
|
+
corpus = self._corpus
|
|
504
|
+
parents = self._parents
|
|
505
|
+
while parents:
|
|
506
|
+
current_parents = [*parents]
|
|
507
|
+
parents = []
|
|
508
|
+
for parent in current_parents:
|
|
509
|
+
parents += parent._parents
|
|
510
|
+
for anc_name, anchors in self._anchorings.items():
|
|
511
|
+
if anc_name not in parent._anchorings:
|
|
512
|
+
parent._anchorings[anc_name] = [*anchors]
|
|
513
|
+
parent_anchors = parent._anchorings[anc_name]
|
|
514
|
+
if anchors[0] < parent_anchors[0]:
|
|
515
|
+
parent_anchors[0] = anchors[0]
|
|
516
|
+
if anc_name == "time" and parent._name == corpus._document:
|
|
517
|
+
if corpus._upperFrameDocument < parent_anchors[0]:
|
|
518
|
+
parent_anchors[0] = corpus._upperFrameDocument
|
|
519
|
+
corpus._upperFrameDocument = parent_anchors[1]
|
|
520
|
+
if anc_name != "location":
|
|
521
|
+
if anchors[1] > parent_anchors[1]:
|
|
522
|
+
parent_anchors[1] = anchors[1]
|
|
523
|
+
continue
|
|
524
|
+
if anchors[1] < parent_anchors[1]:
|
|
525
|
+
parent_anchors[1] = anchors[1]
|
|
526
|
+
if anchors[2] > parent_anchors[2]:
|
|
527
|
+
parent_anchors[2] = anchors[2]
|
|
528
|
+
if anchors[3] > parent_anchors[3]:
|
|
529
|
+
parent_anchors[3] = anchors[3]
|
|
530
|
+
|
|
531
|
+
def clear(self):
|
|
532
|
+
if not self._made:
|
|
533
|
+
return
|
|
534
|
+
# Prepare for deletion: no pointers to other layers/global attributes
|
|
535
|
+
self._parents = []
|
|
536
|
+
self._contains = []
|
|
537
|
+
self._attributes = {}
|
|
538
|
+
|
|
539
|
+
def make(self, clear=False):
|
|
475
540
|
if self._made:
|
|
476
541
|
return
|
|
477
542
|
corpus = self._corpus
|
|
@@ -494,6 +559,9 @@ class Layer:
|
|
|
494
559
|
rows.append(doc_name)
|
|
495
560
|
rows.append(json.dumps(self._media))
|
|
496
561
|
if is_token:
|
|
562
|
+
assert "form" in self._attributes, RuntimeError(
|
|
563
|
+
"Tried to make a token with no form"
|
|
564
|
+
)
|
|
497
565
|
seg_parent = self._find_in_parents(corpus._segment)
|
|
498
566
|
rows.append(seg_parent._id)
|
|
499
567
|
char_low = corpus._char_counter
|
|
@@ -502,35 +570,8 @@ class Layer:
|
|
|
502
570
|
)
|
|
503
571
|
self._anchorings["stream"] = [char_low, corpus._char_counter]
|
|
504
572
|
elif self._contains:
|
|
505
|
-
unset_anchorings = {a for a in ANCHORINGS if not self._anchorings.get(a)}
|
|
506
573
|
for child in self._contains:
|
|
507
574
|
child.make()
|
|
508
|
-
if child._name not in mapping.contains:
|
|
509
|
-
mapping.contains.append(child._name)
|
|
510
|
-
# Anchorings
|
|
511
|
-
for a in unset_anchorings:
|
|
512
|
-
if a not in child._anchorings:
|
|
513
|
-
continue
|
|
514
|
-
child_a = child._anchorings[a]
|
|
515
|
-
if a not in self._anchorings:
|
|
516
|
-
self._anchorings[a] = [*child_a]
|
|
517
|
-
self_a = self._anchorings[a]
|
|
518
|
-
if child_a[0] < self_a[0]:
|
|
519
|
-
self_a[0] = child_a[0]
|
|
520
|
-
if a == "time" and self._name == corpus._document:
|
|
521
|
-
if corpus._upperFrameDocument < self_a[0]:
|
|
522
|
-
self_a[0] = corpus._upperFrameDocument
|
|
523
|
-
corpus._upperFrameDocument = self_a[1]
|
|
524
|
-
if a != "location":
|
|
525
|
-
if child_a[1] > self_a[1]:
|
|
526
|
-
self_a[1] = child_a[1]
|
|
527
|
-
continue
|
|
528
|
-
if child_a[1] < self_a[1]:
|
|
529
|
-
self_a[1] = child_a[1]
|
|
530
|
-
if child_a[2] > self_a[2]:
|
|
531
|
-
self_a[2] = child_a[2]
|
|
532
|
-
if child_a[3] > self_a[3]:
|
|
533
|
-
self_a[3] = child_a[3]
|
|
534
575
|
if is_segment:
|
|
535
576
|
tokens = [
|
|
536
577
|
ch._attributes.values()
|
|
@@ -568,15 +609,27 @@ class Layer:
|
|
|
568
609
|
rows.append(v)
|
|
569
610
|
# Add any new attribute to mapping
|
|
570
611
|
for aname, attr in self._attributes.items():
|
|
612
|
+
atype = attr._type
|
|
571
613
|
if aname in mapping.attributes:
|
|
614
|
+
mattr = mapping.attributes[aname]
|
|
615
|
+
if atype == "text" and mattr["type"] != atype:
|
|
616
|
+
mattr["type"] = "text"
|
|
617
|
+
else:
|
|
618
|
+
try:
|
|
619
|
+
mapping.attributes[aname]["subtype"] = attr._subtype
|
|
620
|
+
except:
|
|
621
|
+
pass
|
|
572
622
|
continue
|
|
573
|
-
atype = attr._type
|
|
574
623
|
mapping.attributes[aname] = {
|
|
575
624
|
"type": atype,
|
|
576
625
|
"nullable": (
|
|
577
626
|
True if mapping.counter > 1 else False
|
|
578
627
|
), # adding a new attribute
|
|
579
628
|
}
|
|
629
|
+
try:
|
|
630
|
+
mapping.attributes[aname]["subtype"] = attr._subtype
|
|
631
|
+
except:
|
|
632
|
+
pass
|
|
580
633
|
if atype == "ref":
|
|
581
634
|
mapping.attributes[aname]["ref"] = attr._ref.lower()
|
|
582
635
|
elif atype in ATYPES_LOOKUP and aname != "meta":
|
|
@@ -656,6 +709,9 @@ class Layer:
|
|
|
656
709
|
rows.append("" if val == None else str(val))
|
|
657
710
|
mapping.csvs["_main"].writerow(rows)
|
|
658
711
|
self._made = True
|
|
712
|
+
self._update_parents_anchors()
|
|
713
|
+
if clear:
|
|
714
|
+
self.clear()
|
|
659
715
|
return self
|
|
660
716
|
|
|
661
717
|
def set_time(self, *args):
|
|
@@ -711,19 +767,25 @@ class Layer:
|
|
|
711
767
|
return self
|
|
712
768
|
|
|
713
769
|
def add(self, *layers: "Layer"):
|
|
714
|
-
assert not self._contains or all(
|
|
715
|
-
|
|
716
|
-
), RuntimeError("All the children of a layer must be of the same type")
|
|
770
|
+
# assert not self._contains or all(
|
|
771
|
+
# l._name == self._contains[0]._name for l in layers
|
|
772
|
+
# ), RuntimeError("All the children of a layer must be of the same type")
|
|
717
773
|
self._contains += layers
|
|
774
|
+
mapping = self._corpus._layers[self._name]
|
|
718
775
|
for layer in layers:
|
|
776
|
+
if layer._name not in mapping.contains:
|
|
777
|
+
mapping.contains.append(layer._name)
|
|
719
778
|
if self not in layer._parents:
|
|
720
779
|
layer._parents.append(self)
|
|
780
|
+
layer._update_parents_anchors()
|
|
721
781
|
return self
|
|
722
782
|
|
|
723
783
|
|
|
724
784
|
class Attribute:
|
|
725
785
|
def __init__(self, layer: Layer, name: str, value: Any = None):
|
|
726
786
|
self._name = name
|
|
787
|
+
if name not in layer._attributes:
|
|
788
|
+
layer._attributes[name] = self
|
|
727
789
|
self._value = value
|
|
728
790
|
self._layer = layer
|
|
729
791
|
self._ref = None
|
|
@@ -739,6 +801,10 @@ class Attribute:
|
|
|
739
801
|
self._value = json.dumps(sorted_dict(value))
|
|
740
802
|
elif isinstance(value, (int, float)):
|
|
741
803
|
atype = "number"
|
|
804
|
+
if isinstance(value, float):
|
|
805
|
+
self._subtype = "float"
|
|
806
|
+
# overwrite to ensure subtype is taken into consideration
|
|
807
|
+
layer._attributes[name] = self
|
|
742
808
|
elif isinstance(value, GlobalAttribute):
|
|
743
809
|
atype = "ref"
|
|
744
810
|
self._ref = value._name
|
|
@@ -747,7 +813,6 @@ class Attribute:
|
|
|
747
813
|
atype = "entity"
|
|
748
814
|
self._ref = value._name
|
|
749
815
|
self._type: str = atype
|
|
750
|
-
layer._attributes[name] = self
|
|
751
816
|
|
|
752
817
|
|
|
753
818
|
class GlobalAttribute:
|