gtfparse 2.2.0__tar.gz → 2.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gtfparse-2.2.0 → gtfparse-2.4.0}/PKG-INFO +1 -1
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse/__init__.py +1 -1
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse/create_missing_features.py +12 -6
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse/read_gtf.py +1 -13
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse.egg-info/PKG-INFO +1 -1
- {gtfparse-2.2.0 → gtfparse-2.4.0}/LICENSE +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/README.md +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse/attribute_parsing.py +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse/parsing_error.py +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse.egg-info/SOURCES.txt +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse.egg-info/dependency_links.txt +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse.egg-info/requires.txt +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/gtfparse.egg-info/top_level.txt +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/pyproject.toml +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/requirements.txt +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/setup.cfg +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/tests/test_create_missing_features.py +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/tests/test_ensembl_gtf.py +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/tests/test_expand_attributes.py +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/tests/test_multiple_values_for_tag_attribute.py +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/tests/test_parse_gtf_lines.py +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/tests/test_read_stringtie_gtf.py +0 -0
- {gtfparse-2.2.0 → gtfparse-2.4.0}/tests/test_refseq_gtf.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.0
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
@@ -49,15 +49,19 @@ def create_missing_features(
|
|
|
49
49
|
missing_value : any
|
|
50
50
|
Which value to fill in for columns that we don't infer values for.
|
|
51
51
|
|
|
52
|
-
Returns original dataframe
|
|
53
|
-
features.
|
|
52
|
+
Returns original dataframe (converted to Pandas if necessary) along with all
|
|
53
|
+
extra rows created for missing features.
|
|
54
54
|
"""
|
|
55
|
+
if hasattr(dataframe, "to_pandas"):
|
|
56
|
+
dataframe = dataframe.to_pandas()
|
|
57
|
+
|
|
55
58
|
extra_dataframes = []
|
|
56
59
|
|
|
57
60
|
existing_features = set(dataframe["feature"])
|
|
58
61
|
existing_columns = set(dataframe.columns)
|
|
59
|
-
|
|
62
|
+
|
|
60
63
|
for (feature_name, groupby_key) in unique_keys.items():
|
|
64
|
+
|
|
61
65
|
if feature_name in existing_features:
|
|
62
66
|
logging.info(
|
|
63
67
|
"Feature '%s' already exists in GTF data" % feature_name)
|
|
@@ -65,9 +69,11 @@ def create_missing_features(
|
|
|
65
69
|
logging.info("Creating rows for missing feature '%s'" % feature_name)
|
|
66
70
|
|
|
67
71
|
# don't include rows where the groupby key was missing
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
72
|
+
missing = pd.Series([
|
|
73
|
+
x is None or x == ""
|
|
74
|
+
for x in dataframe[groupby_key]])
|
|
75
|
+
not_missing = ~missing
|
|
76
|
+
row_groups = dataframe[not_missing].groupby(groupby_key)
|
|
71
77
|
|
|
72
78
|
# Each group corresponds to a unique feature entry for which the
|
|
73
79
|
# other columns may or may not be uniquely defined. Start off by
|
|
@@ -105,22 +105,10 @@ def parse_with_polars_lazy(
|
|
|
105
105
|
null_values=".",
|
|
106
106
|
dtypes=DEFAULT_COLUMN_DTYPES)
|
|
107
107
|
try:
|
|
108
|
-
|
|
109
|
-
df = polars.read_csv(
|
|
108
|
+
df = polars.read_csv(
|
|
110
109
|
filepath_or_buffer,
|
|
111
110
|
new_columns=REQUIRED_COLUMNS,
|
|
112
111
|
**kwargs).lazy()
|
|
113
|
-
elif filepath_or_buffer.endswith(".gz") or filepath_or_buffer.endswith(".gzip"):
|
|
114
|
-
with gzip.open(filepath_or_buffer) as f:
|
|
115
|
-
df = polars.read_csv(
|
|
116
|
-
f,
|
|
117
|
-
new_columns=REQUIRED_COLUMNS,
|
|
118
|
-
**kwargs).lazy()
|
|
119
|
-
else:
|
|
120
|
-
df = polars.scan_csv(
|
|
121
|
-
filepath_or_buffer,
|
|
122
|
-
with_column_names=lambda cols: REQUIRED_COLUMNS,
|
|
123
|
-
**kwargs).lazy()
|
|
124
112
|
except polars.ShapeError:
|
|
125
113
|
raise ParsingError("Wrong number of columns")
|
|
126
114
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: gtfparse
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.4.0
|
|
4
4
|
Summary: Parsing library for extracting data frames of genomic features from GTF files
|
|
5
5
|
Author-email: Alex Rubinsteyn <alex.rubinsteyn@unc.edu>
|
|
6
6
|
Project-URL: Homepage, https://github.com/openvax/gtfparse
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|