bcdata 0.11.0__tar.gz → 0.12.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (29) hide show
  1. {bcdata-0.11.0 → bcdata-0.12.2}/CHANGES.txt +12 -0
  2. {bcdata-0.11.0/bcdata.egg-info → bcdata-0.12.2}/PKG-INFO +2 -2
  3. {bcdata-0.11.0 → bcdata-0.12.2}/README.md +1 -1
  4. bcdata-0.12.2/bcdata/__init__.py +30 -0
  5. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata/bc2pg.py +2 -16
  6. bcdata-0.12.2/bcdata/bcdc.py +134 -0
  7. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata/cli.py +1 -6
  8. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata/wfs.py +2 -3
  9. {bcdata-0.11.0 → bcdata-0.12.2/bcdata.egg-info}/PKG-INFO +2 -2
  10. {bcdata-0.11.0 → bcdata-0.12.2}/tests/test_bc2pg.py +1 -2
  11. {bcdata-0.11.0 → bcdata-0.12.2}/tests/test_bcdc.py +8 -0
  12. bcdata-0.11.0/bcdata/__init__.py +0 -18
  13. bcdata-0.11.0/bcdata/bcdc.py +0 -157
  14. {bcdata-0.11.0 → bcdata-0.12.2}/LICENSE.txt +0 -0
  15. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata/database.py +0 -0
  16. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata/wcs.py +0 -0
  17. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata.egg-info/SOURCES.txt +0 -0
  18. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata.egg-info/dependency_links.txt +0 -0
  19. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata.egg-info/entry_points.txt +0 -0
  20. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata.egg-info/not-zip-safe +0 -0
  21. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata.egg-info/requires.txt +0 -0
  22. {bcdata-0.11.0 → bcdata-0.12.2}/bcdata.egg-info/top_level.txt +0 -0
  23. {bcdata-0.11.0 → bcdata-0.12.2}/pyproject.toml +0 -0
  24. {bcdata-0.11.0 → bcdata-0.12.2}/requirements.txt +0 -0
  25. {bcdata-0.11.0 → bcdata-0.12.2}/setup.cfg +0 -0
  26. {bcdata-0.11.0 → bcdata-0.12.2}/setup.py +0 -0
  27. {bcdata-0.11.0 → bcdata-0.12.2}/tests/test_cli.py +0 -0
  28. {bcdata-0.11.0 → bcdata-0.12.2}/tests/test_wcs.py +0 -0
  29. {bcdata-0.11.0 → bcdata-0.12.2}/tests/test_wfs.py +0 -0
@@ -1,6 +1,18 @@
1
1
  Changes
2
2
  =======
3
3
 
4
+ 0.12.2 (2024-08-15)
5
+ ------------------
6
+ - fix error where incorrect schema returned for datasets with multiple resources (#196)
7
+
8
+ 0.12.1 (2024-08-15)
9
+ ------------------
10
+ - fix syntax error
11
+
12
+ 0.12.0 (2024-08-15)
13
+ ------------------
14
+ - expose primary key database as bcdata.primary_keys and via bcdata info (#193)
15
+
4
16
  0.11.0 (2024-07-29)
5
17
  ------------------
6
18
  - upgrade dependencies
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: bcdata
3
- Version: 0.11.0
3
+ Version: 0.12.2
4
4
  Summary: Python tools for quick access to DataBC geo-data available via WFS
5
5
  Home-page: https://github.com/smnorris/bcdata
6
6
  Author: Simon Norris
@@ -339,7 +339,7 @@ Describe a dataset. Note that if we know the id of a dataset, we can use that ra
339
339
 
340
340
  The JSON output can be manipulated with [jq](https://stedolan.github.io/jq/). For example, to show only the fields available in the dataset:
341
341
 
342
- $ bcdata info bc-airports | jq '.schema.properties'
342
+ $ bcdata info bc-airports | jq '.schema'
343
343
  {
344
344
  "CUSTODIAN_ORG_DESCRIPTION": "string",
345
345
  "BUSINESS_CATEGORY_CLASS": "string",
@@ -305,7 +305,7 @@ Describe a dataset. Note that if we know the id of a dataset, we can use that ra
305
305
 
306
306
  The JSON output can be manipulated with [jq](https://stedolan.github.io/jq/). For example, to show only the fields available in the dataset:
307
307
 
308
- $ bcdata info bc-airports | jq '.schema.properties'
308
+ $ bcdata info bc-airports | jq '.schema'
309
309
  {
310
310
  "CUSTODIAN_ORG_DESCRIPTION": "string",
311
311
  "BUSINESS_CATEGORY_CLASS": "string",
@@ -0,0 +1,30 @@
1
+ import requests
2
+
3
+ from .bc2pg import bc2pg
4
+ from .bcdc import get_table_definition, get_table_name
5
+ from .wcs import get_dem
6
+ from .wfs import (
7
+ define_requests,
8
+ get_count,
9
+ get_data,
10
+ get_features,
11
+ get_sortkey,
12
+ list_tables,
13
+ validate_name,
14
+ )
15
+
16
+ PRIMARY_KEY_DB_URL = (
17
+ "https://raw.githubusercontent.com/smnorris/bcdata/main/data/primary_keys.json"
18
+ )
19
+
20
+ # BCDC does not indicate which column in the schema is the primary key.
21
+ # In this absence, bcdata maintains its own dictionary of {table: primary_key},
22
+ # served via github. Retrieve the dict with this function"""
23
+ response = requests.get(PRIMARY_KEY_DB_URL)
24
+ if response.status_code == 200:
25
+ primary_keys = response.json()
26
+ else:
27
+ raise Exception(f"Failed to download primary key database at {PRIMARY_KEY_DB_URL}")
28
+ primary_keys = {}
29
+
30
+ __version__ = "0.12.2"
@@ -34,19 +34,6 @@ SUPPORTED_TYPES = [
34
34
  ]
35
35
 
36
36
 
37
- def get_primary_keys():
38
- """download primary key data file"""
39
- response = requests.get(bcdata.PRIMARY_KEY_DB_URL)
40
- if response.status_code == 200:
41
- primary_keys = response.json()
42
- else:
43
- log.warning(
44
- f"Failed to download primary key database at {bcdata.PRIMARY_KEY_DB_URL}"
45
- )
46
- primary_keys = {}
47
- return primary_keys
48
-
49
-
50
37
  def bc2pg( # noqa: C901
51
38
  dataset,
52
39
  db_url,
@@ -148,9 +135,8 @@ def bc2pg( # noqa: C901
148
135
  raise ValueError("Geometry type {geometry_type} is not supported")
149
136
 
150
137
  # if primary key is not supplied, use default (if present in list)
151
- primary_keys = get_primary_keys()
152
- if not primary_key and dataset.lower() in primary_keys:
153
- primary_key = primary_keys[dataset.lower()]
138
+ if not primary_key and dataset.lower() in bcdata.primary_keys:
139
+ primary_key = bcdata.primary_keys[dataset.lower()]
154
140
 
155
141
  # fail if specified primary key is not in the table
156
142
  if primary_key and primary_key.upper() not in [
@@ -0,0 +1,134 @@
1
+ import json
2
+ import logging
3
+ from urllib.parse import urlparse
4
+
5
+ import requests
6
+ import stamina
7
+
8
+ import bcdata
9
+
10
+ log = logging.getLogger(__name__)
11
+
12
+ BCDC_API_URL = "https://catalogue.data.gov.bc.ca/api/3/action/"
13
+
14
+
15
+ class ServiceException(Exception):
16
+ pass
17
+
18
+
19
+ @stamina.retry(on=requests.HTTPError, timeout=60)
20
+ def _package_show(package):
21
+ r = requests.get(BCDC_API_URL + "package_show", params={"id": package})
22
+ if r.status_code in [400, 404]:
23
+ log.error(f"HTTP error {r.status_code}")
24
+ log.error(f"Response headers: {r.headers}")
25
+ log.error(f"Response text: {r.text}")
26
+ raise ValueError(f"Dataset {package} not found in DataBC API list")
27
+ if r.status_code in [500, 502, 503, 504]: # presumed serivce error, retry
28
+ log.warning(f"HTTP error: {r.status_code}")
29
+ log.warning(f"Response headers: {r.headers}")
30
+ log.warning(f"Response text: {r.text}")
31
+ r.raise_for_status()
32
+ else:
33
+ log.debug(r.text)
34
+ return r
35
+
36
+
37
+ @stamina.retry(on=requests.HTTPError, timeout=60)
38
+ def _table_definition(table_name):
39
+ r = requests.get(
40
+ BCDC_API_URL + "package_search",
41
+ params={"q": "res_extras_object_name:" + table_name},
42
+ )
43
+ if r.status_code != 200:
44
+ log.warning(r.headers)
45
+ if r.status_code in [400, 401, 404]:
46
+ raise ServiceException(r.text) # presumed request error
47
+ if r.status_code in [500, 502, 503, 504]: # presumed serivce error, retry
48
+ r.raise_for_status()
49
+ return r
50
+
51
+
52
+ def get_table_name(package):
53
+ """Query DataBC API to find WFS table/layer name for given package"""
54
+ package = package.lower() # package names are lowercase
55
+ r = _package_show(package)
56
+ result = r.json()["result"]
57
+ # Because the object_name in the result json is not a 100% reliable key
58
+ # for WFS requests, parse URL in WMS resource(s).
59
+ # Also, some packages may have >1 WFS layer - if this is the case, bail
60
+ # and provide user with a list of layers
61
+ layer_urls = [r["url"] for r in result["resources"] if r["format"] == "wms"]
62
+ layer_names = [urlparse(url).path.split("/")[3] for url in layer_urls]
63
+ if len(layer_names) > 1:
64
+ raise ValueError(
65
+ "Package {} includes more than one WFS resource, specify one of the following: \n{}".format(
66
+ package, "\n".join(layer_names)
67
+ )
68
+ )
69
+ return layer_names[0]
70
+
71
+
72
+ def get_table_definition(table_name):
73
+ """
74
+ Given a table/object name, search BCDC for the first package/resource with a matching "object_name",
75
+ returns dict: {"comments": <>, "notes": <>, "schema": {<schema dict>} }
76
+ """
77
+ # only allow searching for tables present in WFS list
78
+ table_name = table_name.upper()
79
+ if table_name not in bcdata.list_tables():
80
+ raise ValueError(
81
+ f"Only tables available via WFS are supported, {table_name} not found"
82
+ )
83
+
84
+ # search the api for the provided table
85
+ r = _table_definition(table_name)
86
+
87
+ # start with an empty table definition dict
88
+ table_definition = {
89
+ "description": None,
90
+ "comments": None,
91
+ "schema": [],
92
+ "primary_key": None,
93
+ }
94
+
95
+ # if there are no matching results, let the user know
96
+ if r.json()["result"]["count"] == 0:
97
+ log.warning(
98
+ f"BC Data Catalouge API search provides no results for: {table_name}"
99
+ )
100
+ else:
101
+ # iterate through results of search (packages)
102
+ for result in r.json()["result"]["results"]:
103
+ # description is at top level, same for all resources
104
+ table_definition["description"] = result["notes"]
105
+ # iterate through resources associated with each package
106
+ for resource in result["resources"]:
107
+ # only examine geographic resources with object name key
108
+ if (
109
+ "object_name" in resource.keys()
110
+ and resource["bcdc_type"] == "geographic"
111
+ ):
112
+ # confirm that object name matches table name and schema is present
113
+ if (
114
+ resource["object_name"] == table_name
115
+ and "details" in resource.keys()
116
+ and resource["details"] != ""
117
+ ):
118
+ table_definition["schema"] = json.loads(resource["details"])
119
+ # look for comments only if details/schema was found
120
+ if "object_table_comments" in resource.keys():
121
+ table_definition["comments"] = resource[
122
+ "object_table_comments"
123
+ ]
124
+
125
+ if not table_definition["schema"]:
126
+ log.warning(
127
+ f"BC Data Catalouge API search provides no schema for: {table_name}"
128
+ )
129
+
130
+ # add primary key if present in bcdata.primary_keys
131
+ if table_name.lower() in bcdata.primary_keys:
132
+ table_definition["primary_key"] = bcdata.primary_keys[table_name.lower()]
133
+
134
+ return table_definition
@@ -131,14 +131,9 @@ def info(dataset, indent, meta_member, verbose, quiet):
131
131
  verbosity = verbose - quiet
132
132
  configure_logging(verbosity)
133
133
  dataset = bcdata.validate_name(dataset)
134
- info = {}
134
+ info = bcdata.get_table_definition(dataset)
135
135
  info["name"] = dataset
136
136
  info["count"] = bcdata.get_count(dataset)
137
- table_definition = bcdata.get_table_definition(dataset)
138
- info["description"] = table_definition["description"]
139
- info["table_comments"] = table_definition["comments"]
140
- info["schema"] = table_definition["schema"]
141
-
142
137
  if meta_member:
143
138
  click.echo(info[meta_member])
144
139
  else:
@@ -223,9 +223,8 @@ class BCWFS(object):
223
223
  """Check data for unique columns available for sorting paged requests"""
224
224
  columns = list(self.get_schema(table)["properties"].keys())
225
225
  # use known primary key if it is present in the bcdata repository
226
- known_primary_keys = bcdata.get_primary_keys()
227
- if table.lower() in known_primary_keys:
228
- return known_primary_keys[table.lower()].upper()
226
+ if table.lower() in bcdata.primary_keys:
227
+ return bcdata.primary_keys[table.lower()].upper()
229
228
  # if pk not known, use OBJECTID as default sort key when present
230
229
  elif "OBJECTID" in columns:
231
230
  return "OBJECTID"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: bcdata
3
- Version: 0.11.0
3
+ Version: 0.12.2
4
4
  Summary: Python tools for quick access to DataBC geo-data available via WFS
5
5
  Home-page: https://github.com/smnorris/bcdata
6
6
  Author: Simon Norris
@@ -339,7 +339,7 @@ Describe a dataset. Note that if we know the id of a dataset, we can use that ra
339
339
 
340
340
  The JSON output can be manipulated with [jq](https://stedolan.github.io/jq/). For example, to show only the fields available in the dataset:
341
341
 
342
- $ bcdata info bc-airports | jq '.schema.properties'
342
+ $ bcdata info bc-airports | jq '.schema'
343
343
  {
344
344
  "CUSTODIAN_ORG_DESCRIPTION": "string",
345
345
  "BUSINESS_CATEGORY_CLASS": "string",
@@ -126,8 +126,7 @@ def test_bc2pg_primary_key():
126
126
 
127
127
 
128
128
  def test_bc2pg_get_primary_keys():
129
- primary_keys = bcdata.get_primary_keys()
130
- assert primary_keys[ASSESSMENTS_TABLE] == "stream_crossing_id"
129
+ assert bcdata.primary_keys[ASSESSMENTS_TABLE] == "stream_crossing_id"
131
130
 
132
131
 
133
132
  def test_bc2pg_primary_key_default():
@@ -2,6 +2,7 @@ import json
2
2
 
3
3
  import pytest
4
4
 
5
+ import bcdata
5
6
  from bcdata import bcdc
6
7
 
7
8
  AIRPORTS_PACKAGE = "bc-airports"
@@ -40,6 +41,13 @@ def test_get_table_definition_format_multi():
40
41
  assert table_definition["description"]
41
42
  assert table_definition["comments"]
42
43
  assert table_definition["schema"]
44
+ columns = [c["column_name"] for c in table_definition["schema"]]
45
+ assert (
46
+ bcdata.primary_keys[
47
+ "whse_forest_vegetation.ogsr_priority_def_area_cur_sp"
48
+ ].upper()
49
+ in columns
50
+ )
43
51
 
44
52
 
45
53
  def test_get_table_definition_format_multi_nopreview():
@@ -1,18 +0,0 @@
1
- from .bc2pg import bc2pg, get_primary_keys
2
- from .bcdc import get_table_definition, get_table_name
3
- from .wcs import get_dem
4
- from .wfs import (
5
- define_requests,
6
- get_count,
7
- get_data,
8
- get_features,
9
- get_sortkey,
10
- list_tables,
11
- validate_name,
12
- )
13
-
14
- PRIMARY_KEY_DB_URL = (
15
- "https://raw.githubusercontent.com/smnorris/bcdata/main/data/primary_keys.json"
16
- )
17
-
18
- __version__ = "0.11.0"
@@ -1,157 +0,0 @@
1
- import json
2
- import logging
3
- from urllib.parse import urlparse
4
-
5
- import requests
6
- import stamina
7
-
8
- import bcdata
9
-
10
- log = logging.getLogger(__name__)
11
-
12
- BCDC_API_URL = "https://catalogue.data.gov.bc.ca/api/3/action/"
13
-
14
-
15
- class ServiceException(Exception):
16
- pass
17
-
18
-
19
- @stamina.retry(on=requests.HTTPError, timeout=60)
20
- def _package_show(package):
21
- r = requests.get(BCDC_API_URL + "package_show", params={"id": package})
22
- if r.status_code in [400, 404]:
23
- log.error(f"HTTP error {r.status_code}")
24
- log.error(f"Response headers: {r.headers}")
25
- log.error(f"Response text: {r.text}")
26
- raise ValueError(f"Dataset {package} not found in DataBC API list")
27
- if r.status_code in [500, 502, 503, 504]: # presumed serivce error, retry
28
- log.warning(f"HTTP error: {r.status_code}")
29
- log.warning(f"Response headers: {r.headers}")
30
- log.warning(f"Response text: {r.text}")
31
- r.raise_for_status()
32
- else:
33
- log.debug(r.text)
34
- return r
35
-
36
-
37
- @stamina.retry(on=requests.HTTPError, timeout=60)
38
- def _table_definition(table_name):
39
- r = requests.get(BCDC_API_URL + "package_search", params={"q": table_name})
40
- if r.status_code != 200:
41
- log.warning(r.headers)
42
- if r.status_code in [400, 401, 404]:
43
- raise ServiceException(r.text) # presumed request error
44
- if r.status_code in [500, 502, 503, 504]: # presumed serivce error, retry
45
- r.raise_for_status()
46
- return r
47
-
48
-
49
- def get_table_name(package):
50
- """Query DataBC API to find WFS table/layer name for given package"""
51
- package = package.lower() # package names are lowercase
52
- r = _package_show(package)
53
- result = r.json()["result"]
54
- # Because the object_name in the result json is not a 100% reliable key
55
- # for WFS requests, parse URL in WMS resource(s).
56
- # Also, some packages may have >1 WFS layer - if this is the case, bail
57
- # and provide user with a list of layers
58
- layer_urls = [r["url"] for r in result["resources"] if r["format"] == "wms"]
59
- layer_names = [urlparse(url).path.split("/")[3] for url in layer_urls]
60
- if len(layer_names) > 1:
61
- raise ValueError(
62
- "Package {} includes more than one WFS resource, specify one of the following: \n{}".format(
63
- package, "\n".join(layer_names)
64
- )
65
- )
66
- return layer_names[0]
67
-
68
-
69
- def get_table_definition(table_name): # noqa: C901
70
- """
71
- Given a table/object name, search BCDC for the first package/resource with a matching "object_name",
72
- returns dict: {"comments": <>, "notes": <>, "schema": {<schema dict>} }
73
- """
74
- # only allow searching for tables present in WFS list
75
- table_name = table_name.upper()
76
- if table_name not in bcdata.list_tables():
77
- raise ValueError(
78
- f"Only tables available via WFS are supported, {table_name} not found"
79
- )
80
- # search the api for the provided table
81
- r = _table_definition(table_name)
82
- # if there are no matching results, let the user know
83
- if r.json()["result"]["count"] == 0:
84
- log.warning(
85
- f"BC Data Catalouge API search provides no results for: {table_name}"
86
- )
87
- return []
88
- else:
89
- matches = []
90
- # iterate through results of search (packages)
91
- for result in r.json()["result"]["results"]:
92
- notes = result["notes"]
93
- # iterate through resources associated with each package
94
- for resource in result["resources"]:
95
- # where to find schema details depends on format type
96
- if resource["format"] == "wms":
97
- if urlparse(resource["url"]).path.split("/")[3] == table_name:
98
- if "object_table_comments" in resource.keys():
99
- table_comments = resource["object_table_comments"]
100
- else:
101
- table_comments = None
102
- # only add to matches if schema details found
103
- if "details" in resource.keys() and resource["details"] != "":
104
- table_details = resource["details"]
105
- matches.append((notes, table_comments, table_details))
106
- log.debug(resource)
107
- # oracle sde format type
108
- if resource["format"] == "oracle_sde":
109
- if resource["object_name"] == table_name:
110
- if "object_table_comments" in resource.keys():
111
- table_comments = resource["object_table_comments"]
112
- else:
113
- table_comments = None
114
- # only add to matches if schema details found
115
- if "details" in resource.keys() and resource["details"] != "":
116
- table_details = resource["details"]
117
- matches.append((notes, table_comments, table_details))
118
- log.debug(resource)
119
-
120
- # multiple format resource
121
- elif resource["format"] == "multiple":
122
- # if multiple format, check for table name match in this location
123
- if resource["preview_info"]:
124
- # check that layer_name key is present
125
- if "layer_name" in json.loads(resource["preview_info"]):
126
- # then check if it matches the table name
127
- if (
128
- json.loads(resource["preview_info"])["layer_name"]
129
- == table_name
130
- ):
131
- if "object_table_comments" in resource.keys():
132
- table_comments = resource["object_table_comments"]
133
- else:
134
- table_comments = None
135
- # only add to matches if schema details found
136
- if (
137
- "details" in resource.keys()
138
- and resource["details"] != ""
139
- ):
140
- table_details = resource["details"]
141
- matches.append(
142
- (notes, table_comments, table_details)
143
- )
144
- log.debug(resource)
145
-
146
- # uniquify the result
147
- if len(matches) > 0:
148
- matched = list(set(matches))[0]
149
- return {
150
- "description": matched[0], # notes=description
151
- "comments": matched[1],
152
- "schema": json.loads(matched[2]),
153
- }
154
- else:
155
- raise ValueError(
156
- f"BCDC search for {table_name} does not return a table schema"
157
- )
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes