fielddash 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fielddash/__init__.py +17 -0
- fielddash/access.py +73 -0
- fielddash/app.py +24 -0
- fielddash/cli.py +97 -0
- fielddash/core/__init__.py +0 -0
- fielddash/core/config.py +115 -0
- fielddash/core/context.py +14 -0
- fielddash/core/env.py +20 -0
- fielddash/core/loader.py +47 -0
- fielddash/core/model.py +84 -0
- fielddash/core/normalize.py +104 -0
- fielddash/core/registry.py +27 -0
- fielddash/core/schema.py +123 -0
- fielddash/scaffold.py +149 -0
- fielddash/sources/__init__.py +0 -0
- fielddash/sources/base.py +62 -0
- fielddash/sources/epicollect.py +196 -0
- fielddash/sources/json_file.py +35 -0
- fielddash/ui/__init__.py +0 -0
- fielddash/ui/charts.py +202 -0
- fielddash/ui/compat.py +23 -0
- fielddash/ui/filters.py +90 -0
- fielddash/ui/maps.py +56 -0
- fielddash/views/__init__.py +0 -0
- fielddash/views/data.py +36 -0
- fielddash/views/highlights.py +37 -0
- fielddash/views/overview.py +41 -0
- fielddash/views/questions.py +62 -0
- fielddash/web.py +93 -0
- fielddash-0.1.0.dist-info/METADATA +267 -0
- fielddash-0.1.0.dist-info/RECORD +35 -0
- fielddash-0.1.0.dist-info/WHEEL +5 -0
- fielddash-0.1.0.dist-info/entry_points.txt +2 -0
- fielddash-0.1.0.dist-info/licenses/LICENSE +21 -0
- fielddash-0.1.0.dist-info/top_level.txt +1 -0
fielddash/__init__.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""fielddash: schema-driven dashboards for field data collection (Epicollect5)."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.1.0"
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def dashboard(target=".", *, configure_page: bool = True):
|
|
7
|
+
"""Renders the dashboard for a project (.yaml) or directory of projects in a Streamlit script."""
|
|
8
|
+
from fielddash.web import dashboard as _dashboard # imports streamlit only when called
|
|
9
|
+
|
|
10
|
+
return _dashboard(target, configure_page=configure_page)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def require_access():
|
|
14
|
+
"""Optional PIN gate: stops the Streamlit script on a login screen unless FIELD_ACCESS_PIN matches."""
|
|
15
|
+
from fielddash.access import require_access as _require_access # imports streamlit only when called
|
|
16
|
+
|
|
17
|
+
return _require_access()
|
fielddash/access.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Optional PIN gate for dashboards shared with a field team (Streamlit).
|
|
2
|
+
|
|
3
|
+
import fielddash
|
|
4
|
+
fielddash.require_access() # no-op unless FIELD_ACCESS_PIN is set (env, .env or Streamlit secrets)
|
|
5
|
+
fielddash.dashboard("project.yaml", configure_page=False)
|
|
6
|
+
|
|
7
|
+
Team members enter the code on a login screen or open a link ending in `?token=<code>`.
|
|
8
|
+
"""
|
|
9
|
+
import hmac
|
|
10
|
+
|
|
11
|
+
import streamlit as st
|
|
12
|
+
|
|
13
|
+
from fielddash.core.env import getenv
|
|
14
|
+
|
|
15
|
+
PIN_VARS = ("FIELD_ACCESS_PIN", "FIELD_ACCESS_TOKEN", "ACCESS_PIN")
|
|
16
|
+
URL_PARAMS = ("token", "pin", "key")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def required_pin():
|
|
20
|
+
"""The configured access code, or None when the dashboard is open to everyone."""
|
|
21
|
+
for name in PIN_VARS:
|
|
22
|
+
value = getenv(name)
|
|
23
|
+
if value:
|
|
24
|
+
return str(value).strip()
|
|
25
|
+
return None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _matches(given, expected) -> bool:
|
|
29
|
+
return hmac.compare_digest(str(given).strip().encode(), str(expected).encode())
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _url_token():
|
|
33
|
+
for name in URL_PARAMS:
|
|
34
|
+
value = st.query_params.get(name)
|
|
35
|
+
if value:
|
|
36
|
+
return value
|
|
37
|
+
return None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _sidebar_logout():
|
|
41
|
+
st.sidebar.caption("🟢 Access granted (field team)")
|
|
42
|
+
if st.sidebar.button("End session", key="logout_btn"):
|
|
43
|
+
st.session_state["authenticated"] = False
|
|
44
|
+
for name in URL_PARAMS: # otherwise the link would log the user straight back in
|
|
45
|
+
if name in st.query_params:
|
|
46
|
+
del st.query_params[name]
|
|
47
|
+
st.rerun()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def require_access() -> None:
|
|
51
|
+
"""Stops the script on a login screen unless the visitor has the access code."""
|
|
52
|
+
pin = required_pin()
|
|
53
|
+
if not pin:
|
|
54
|
+
return
|
|
55
|
+
|
|
56
|
+
token = _url_token()
|
|
57
|
+
if st.session_state.get("authenticated") or (token and _matches(token, pin)):
|
|
58
|
+
st.session_state["authenticated"] = True
|
|
59
|
+
_sidebar_logout()
|
|
60
|
+
return
|
|
61
|
+
|
|
62
|
+
_, center, _ = st.columns([1, 2, 1])
|
|
63
|
+
with center:
|
|
64
|
+
st.markdown("## 🔐 Restricted area — field team")
|
|
65
|
+
st.info("This dashboard contains field-work data. Enter the access code provided by the research coordination.")
|
|
66
|
+
with st.form("login_form"):
|
|
67
|
+
entered = st.text_input("Access code / PIN:", type="password")
|
|
68
|
+
if st.form_submit_button("Enter dashboard"):
|
|
69
|
+
if entered and _matches(entered, pin):
|
|
70
|
+
st.session_state["authenticated"] = True
|
|
71
|
+
st.rerun()
|
|
72
|
+
st.error("Wrong access code. Check with the responsible team.")
|
|
73
|
+
st.stop()
|
fielddash/app.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Streamlit script used by `fielddash run`.
|
|
2
|
+
|
|
3
|
+
fielddash run project.yaml # single project
|
|
4
|
+
fielddash run dir/ # choose between .yaml projects in sidebar
|
|
5
|
+
|
|
6
|
+
To deploy (e.g. Streamlit Community Cloud), use `fielddash.dashboard(...)`
|
|
7
|
+
in your project's script.
|
|
8
|
+
"""
|
|
9
|
+
import argparse
|
|
10
|
+
import os
|
|
11
|
+
|
|
12
|
+
from fielddash.web import dashboard
|
|
13
|
+
|
|
14
|
+
ENV_CONFIG = "FIELDDASH_CONFIG"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _config_arg() -> str:
|
|
18
|
+
parser = argparse.ArgumentParser()
|
|
19
|
+
parser.add_argument("--config")
|
|
20
|
+
args, _ = parser.parse_known_args()
|
|
21
|
+
return args.config or os.environ.get(ENV_CONFIG) or "."
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
dashboard(_config_arg())
|
fielddash/cli.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Command line interface for fielddash.
|
|
2
|
+
|
|
3
|
+
fielddash run project.yaml [streamlit options, e.g. --server.port 8600]
|
|
4
|
+
fielddash run dir/ # select project in sidebar
|
|
5
|
+
fielddash fields project.yaml # inspect form fields to write config
|
|
6
|
+
fielddash init my-survey # scaffold a new project folder
|
|
7
|
+
"""
|
|
8
|
+
import argparse
|
|
9
|
+
import subprocess
|
|
10
|
+
import sys
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from fielddash.core.config import load_config
|
|
14
|
+
from fielddash.core.loader import bootstrap, load_dataset
|
|
15
|
+
from fielddash.scaffold import init_project
|
|
16
|
+
|
|
17
|
+
APP = Path(__file__).resolve().parent / "app.py"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def run(target: str, streamlit_args: list) -> int:
|
|
21
|
+
path = Path(target).resolve()
|
|
22
|
+
if not path.exists():
|
|
23
|
+
sys.exit(f"Not found: {path}")
|
|
24
|
+
command = [sys.executable, "-m", "streamlit", "run", str(APP), *streamlit_args, "--", "--config", str(path)]
|
|
25
|
+
return subprocess.call(command)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def fields(target: str) -> int:
|
|
29
|
+
config = load_config(target)
|
|
30
|
+
bootstrap(config)
|
|
31
|
+
dataset = load_dataset(config)
|
|
32
|
+
print(f"{dataset.title}: {len(dataset.df)} responses\n")
|
|
33
|
+
print(f"{'ref (suffix)':<14}{'alias':<20}{'type':<12}{'column':<22}question")
|
|
34
|
+
for f in dataset.fields:
|
|
35
|
+
ref = f.ref if f.system else f.ref[-6:]
|
|
36
|
+
print(f"{ref:<14}{f.alias or '':<20}{f.type:<12}{f.column:<22}{f.label[:70]}")
|
|
37
|
+
for warning in dataset.warnings:
|
|
38
|
+
print(f"⚠ {warning}")
|
|
39
|
+
return 0
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def init(args) -> int:
|
|
43
|
+
directory = Path(args.directory)
|
|
44
|
+
created, skipped = init_project(directory, args.name, args.source, args.deploy, args.force)
|
|
45
|
+
print(f"{directory.resolve()}")
|
|
46
|
+
for relative in created:
|
|
47
|
+
print(f" + {relative}")
|
|
48
|
+
for relative in skipped:
|
|
49
|
+
print(f" = {relative} (already exists, use --force to overwrite)")
|
|
50
|
+
steps = []
|
|
51
|
+
if directory != Path("."):
|
|
52
|
+
steps.append(f"cd {directory}")
|
|
53
|
+
steps.append("python -m venv .venv && source .venv/bin/activate # Windows: .venv\\Scripts\\activate (skip if you already use an environment)")
|
|
54
|
+
steps.append("pip install -r requirements.txt")
|
|
55
|
+
if args.source == "epicollect":
|
|
56
|
+
steps.append("cp .env.example .env # fill in the project slug and credentials")
|
|
57
|
+
else:
|
|
58
|
+
steps.append("save the form schema and entries to data/form.json and data/entries.json")
|
|
59
|
+
steps += ["fielddash fields project.yaml # list fields, then edit project.yaml", "fielddash run project.yaml"]
|
|
60
|
+
print("\nNext steps:")
|
|
61
|
+
for number, step in enumerate(steps, 1):
|
|
62
|
+
print(f" {number}. {step}")
|
|
63
|
+
return 0
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def main(argv=None):
|
|
67
|
+
parser = argparse.ArgumentParser(prog="fielddash", description="Schema-driven dashboards for field data collection.")
|
|
68
|
+
commands = parser.add_subparsers(dest="command", required=True)
|
|
69
|
+
|
|
70
|
+
run_parser = commands.add_parser("run", help="launch dashboard for a project (.yaml) or project directory")
|
|
71
|
+
run_parser.add_argument("project", nargs="?", default=".")
|
|
72
|
+
|
|
73
|
+
fields_parser = commands.add_parser("fields", help="list form fields for a project")
|
|
74
|
+
fields_parser.add_argument("project")
|
|
75
|
+
|
|
76
|
+
init_parser = commands.add_parser("init", help="create a new project folder (project.yaml, .env.example, ...)")
|
|
77
|
+
init_parser.add_argument("directory", nargs="?", default=".", help="folder to create/use (default: current)")
|
|
78
|
+
init_parser.add_argument("--name", help="project name, used for title and variable names (default: folder name)")
|
|
79
|
+
init_parser.add_argument("--source", choices=["epicollect", "json"], default="epicollect")
|
|
80
|
+
init_parser.add_argument("--deploy", action="store_true", help="also create streamlit_app.py, requirements.txt and secrets example")
|
|
81
|
+
init_parser.add_argument("--force", action="store_true", help="overwrite existing files")
|
|
82
|
+
|
|
83
|
+
args, extra = parser.parse_known_args(argv)
|
|
84
|
+
if args.command == "run":
|
|
85
|
+
return run(args.project, extra)
|
|
86
|
+
if extra:
|
|
87
|
+
parser.error(f"unrecognized arguments: {' '.join(extra)}")
|
|
88
|
+
if args.command == "init":
|
|
89
|
+
return init(args)
|
|
90
|
+
try:
|
|
91
|
+
return fields(args.project)
|
|
92
|
+
except (OSError, KeyError, ValueError, RuntimeError) as error:
|
|
93
|
+
sys.exit(f"fielddash: {error}")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
if __name__ == "__main__":
|
|
97
|
+
sys.exit(main())
|
|
File without changes
|
fielddash/core/config.py
ADDED
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
import re
|
|
2
|
+
import warnings
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
import yaml
|
|
7
|
+
from dotenv import load_dotenv
|
|
8
|
+
|
|
9
|
+
from fielddash.core.env import getenv
|
|
10
|
+
|
|
11
|
+
_ENV = re.compile(r"\$\{([A-Za-z0-9_]+)\}")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _expand(value):
|
|
15
|
+
if isinstance(value, str):
|
|
16
|
+
def replace(match):
|
|
17
|
+
name = match.group(1)
|
|
18
|
+
value = getenv(name)
|
|
19
|
+
if value is None:
|
|
20
|
+
raise KeyError(f"Variable not defined in .env or Streamlit secrets: {name}")
|
|
21
|
+
return value
|
|
22
|
+
return _ENV.sub(replace, value)
|
|
23
|
+
if isinstance(value, list):
|
|
24
|
+
return [_expand(v) for v in value]
|
|
25
|
+
if isinstance(value, dict):
|
|
26
|
+
return {k: _expand(v) for k, v in value.items()}
|
|
27
|
+
return value
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# Portuguese keys from earlier versions: still accepted (with a FutureWarning), use English keys.
|
|
31
|
+
LEGACY_KEYS = {
|
|
32
|
+
"titulo": "title",
|
|
33
|
+
"subtitulo": "subtitle",
|
|
34
|
+
"fonte": "source",
|
|
35
|
+
"campos": "fields",
|
|
36
|
+
"ignorar": "ignore",
|
|
37
|
+
"tipos": "types",
|
|
38
|
+
"filtros": "filters",
|
|
39
|
+
"destaques": "highlights",
|
|
40
|
+
"secoes": "sections",
|
|
41
|
+
"mapa": "map",
|
|
42
|
+
"extensoes": "extensions",
|
|
43
|
+
"fuso": "timezone",
|
|
44
|
+
"cache_minutos": "cache_minutes",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
LEGACY_SOURCE_KEYS = {
|
|
48
|
+
"tipo": "type",
|
|
49
|
+
"projeto": "project",
|
|
50
|
+
"credenciais": "credentials",
|
|
51
|
+
"dados": "data",
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@dataclass
|
|
56
|
+
class Config:
|
|
57
|
+
path: Path
|
|
58
|
+
title: str
|
|
59
|
+
source: dict
|
|
60
|
+
subtitle: str = ""
|
|
61
|
+
fields: dict = field(default_factory=dict) # alias -> ref / column / question
|
|
62
|
+
ignore: list = field(default_factory=list)
|
|
63
|
+
types: dict = field(default_factory=dict) # field -> forced type (e.g. neighborhood: category)
|
|
64
|
+
filters: list = field(default_factory=list)
|
|
65
|
+
highlights: list = field(default_factory=list)
|
|
66
|
+
sections: list = field(default_factory=list)
|
|
67
|
+
map: dict = field(default_factory=dict)
|
|
68
|
+
extensions: list = field(default_factory=list) # .py files or directories with project pages/sources
|
|
69
|
+
timezone: str = "America/Fortaleza"
|
|
70
|
+
cache_minutes: int = 5
|
|
71
|
+
|
|
72
|
+
@property
|
|
73
|
+
def base_dir(self) -> Path:
|
|
74
|
+
return self.path.parent
|
|
75
|
+
|
|
76
|
+
def resolve(self, relative: str) -> Path:
|
|
77
|
+
return (self.base_dir / relative).resolve()
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def load_config(path) -> Config:
|
|
81
|
+
path = Path(path).resolve()
|
|
82
|
+
# .env in current directory and in project directory
|
|
83
|
+
load_dotenv(Path.cwd() / ".env")
|
|
84
|
+
load_dotenv(path.parent / ".env")
|
|
85
|
+
with open(path, encoding="utf-8") as f:
|
|
86
|
+
raw = yaml.safe_load(f) or {}
|
|
87
|
+
|
|
88
|
+
normalized, legacy = {}, []
|
|
89
|
+
for key, value in raw.items():
|
|
90
|
+
canonical = LEGACY_KEYS.get(key, key)
|
|
91
|
+
if canonical != key:
|
|
92
|
+
legacy.append(f"{key} -> {canonical}")
|
|
93
|
+
normalized[canonical] = value
|
|
94
|
+
|
|
95
|
+
source = {}
|
|
96
|
+
for key, value in _expand(normalized.get("source") or {}).items():
|
|
97
|
+
canonical = LEGACY_SOURCE_KEYS.get(key, key)
|
|
98
|
+
if canonical != key:
|
|
99
|
+
legacy.append(f"source.{key} -> source.{canonical}")
|
|
100
|
+
source[canonical] = value
|
|
101
|
+
normalized["source"] = source
|
|
102
|
+
|
|
103
|
+
if legacy:
|
|
104
|
+
warnings.warn(f"{path.name}: Portuguese config keys are deprecated, rename: {', '.join(legacy)}", FutureWarning, stacklevel=2)
|
|
105
|
+
|
|
106
|
+
if "type" not in source:
|
|
107
|
+
raise ValueError(f"{path.name}: 'source.type' is required")
|
|
108
|
+
|
|
109
|
+
known = Config.__dataclass_fields__.keys() - {"path"}
|
|
110
|
+
unknown = set(normalized) - known
|
|
111
|
+
if unknown:
|
|
112
|
+
raise ValueError(f"{path.name}: unknown keys: {', '.join(sorted(unknown))}")
|
|
113
|
+
|
|
114
|
+
normalized.setdefault("title", path.stem)
|
|
115
|
+
return Config(path=path, **{k: v for k, v in normalized.items() if v is not None})
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
from fielddash.core.config import Config
|
|
6
|
+
from fielddash.core.model import Dataset
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class PageContext:
|
|
11
|
+
config: Config
|
|
12
|
+
dataset: Dataset # complete data + schema
|
|
13
|
+
df: pd.DataFrame # data after sidebar filters
|
|
14
|
+
filters: list # active filter labels
|
fielddash/core/env.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""Configuration variables: environment (.env) or Streamlit Secrets."""
|
|
2
|
+
import os
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def getenv(name: str) -> str | None:
|
|
6
|
+
"""Retrieves value of `name` from environment (including .env) or `st.secrets`.
|
|
7
|
+
|
|
8
|
+
`st.secrets` comes from `.streamlit/secrets.toml` or Streamlit Community Cloud
|
|
9
|
+
dashboard Secrets, where there is no local .env file.
|
|
10
|
+
"""
|
|
11
|
+
value = os.environ.get(name)
|
|
12
|
+
if value:
|
|
13
|
+
return value
|
|
14
|
+
try:
|
|
15
|
+
import streamlit as st
|
|
16
|
+
|
|
17
|
+
value = st.secrets.get(name)
|
|
18
|
+
except Exception: # no secrets configured or outside of Streamlit runtime
|
|
19
|
+
return None
|
|
20
|
+
return str(value) if value is not None else None
|
fielddash/core/loader.py
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
import importlib
|
|
2
|
+
import importlib.util
|
|
3
|
+
import pkgutil
|
|
4
|
+
import sys
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from fielddash.core.config import Config
|
|
8
|
+
from fielddash.core.model import Dataset
|
|
9
|
+
from fielddash.core.registry import SOURCE_REGISTRY
|
|
10
|
+
|
|
11
|
+
_loaded_extensions = set()
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def bootstrap(config: Config | None = None):
|
|
15
|
+
"""Registers sources and views from the package, and project extensions if config is given."""
|
|
16
|
+
for package in ("fielddash.sources", "fielddash.views"):
|
|
17
|
+
module = importlib.import_module(package)
|
|
18
|
+
for info in pkgutil.iter_modules(module.__path__):
|
|
19
|
+
importlib.import_module(f"{package}.{info.name}")
|
|
20
|
+
|
|
21
|
+
if config is not None:
|
|
22
|
+
for entry in config.extensions:
|
|
23
|
+
path = config.resolve(entry)
|
|
24
|
+
for file in (sorted(path.glob("*.py")) if path.is_dir() else [path]):
|
|
25
|
+
_load_extension(file)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _load_extension(file: Path):
|
|
29
|
+
"""Imports a project module (e.g. project-specific @page) once."""
|
|
30
|
+
file = file.resolve()
|
|
31
|
+
if file in _loaded_extensions or file.name.startswith("_"):
|
|
32
|
+
return
|
|
33
|
+
if not file.exists():
|
|
34
|
+
raise FileNotFoundError(f"Extension not found: {file}")
|
|
35
|
+
name = f"fielddash_ext.{file.stem}"
|
|
36
|
+
spec = importlib.util.spec_from_file_location(name, file)
|
|
37
|
+
module = importlib.util.module_from_spec(spec)
|
|
38
|
+
sys.modules[name] = module
|
|
39
|
+
spec.loader.exec_module(module)
|
|
40
|
+
_loaded_extensions.add(file)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def load_dataset(config: Config) -> Dataset:
|
|
44
|
+
kind = config.source.get("type")
|
|
45
|
+
if kind not in SOURCE_REGISTRY:
|
|
46
|
+
raise ValueError(f"Unknown data source: {kind!r} (available: {', '.join(SOURCE_REGISTRY)})")
|
|
47
|
+
return SOURCE_REGISTRY[kind](config).load()
|
fielddash/core/model.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
from dataclasses import dataclass, field as dc_field
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
# Grouping of Epicollect5 types by how they are handled in the dashboard
|
|
6
|
+
CATEGORICAL = {"radio", "dropdown", "searchsingle", "category"}
|
|
7
|
+
MULTI = {"checkbox", "searchmultiple"}
|
|
8
|
+
NUMERIC = {"integer", "decimal"}
|
|
9
|
+
DATE = {"date", "datetime"}
|
|
10
|
+
LOCATION = {"location"}
|
|
11
|
+
TEXT = {"text", "textarea", "phone", "barcode", "time"}
|
|
12
|
+
MEDIA = {"photo", "audio", "video"}
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class Field:
|
|
17
|
+
ref: str
|
|
18
|
+
column: str
|
|
19
|
+
label: str
|
|
20
|
+
type: str
|
|
21
|
+
options: list = dc_field(default_factory=list)
|
|
22
|
+
group: str | None = None
|
|
23
|
+
alias: str | None = None
|
|
24
|
+
system: bool = False # Epicollect system fields (created_at, created_by...)
|
|
25
|
+
question: str = "" # original question text (used to match columns)
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def kind(self) -> str:
|
|
29
|
+
"""Category used by the UI: categorical, multi, numeric, date, location, text, media or other."""
|
|
30
|
+
for kind, types in (
|
|
31
|
+
("categorical", CATEGORICAL),
|
|
32
|
+
("multi", MULTI),
|
|
33
|
+
("numeric", NUMERIC),
|
|
34
|
+
("date", DATE),
|
|
35
|
+
("location", LOCATION),
|
|
36
|
+
("text", TEXT),
|
|
37
|
+
("media", MEDIA),
|
|
38
|
+
):
|
|
39
|
+
if self.type in types:
|
|
40
|
+
return kind
|
|
41
|
+
return "other"
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def lat(self) -> str:
|
|
45
|
+
return f"{self.column}__lat"
|
|
46
|
+
|
|
47
|
+
@property
|
|
48
|
+
def lon(self) -> str:
|
|
49
|
+
return f"{self.column}__lon"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class Dataset:
|
|
54
|
+
df: pd.DataFrame
|
|
55
|
+
fields: list
|
|
56
|
+
title: str = ""
|
|
57
|
+
warnings: list = dc_field(default_factory=list)
|
|
58
|
+
|
|
59
|
+
def field(self, key: str) -> Field:
|
|
60
|
+
"""Looks up a field by alias, ref (full or suffix), column, or question text."""
|
|
61
|
+
found = self.find(key)
|
|
62
|
+
if found is None:
|
|
63
|
+
raise KeyError(f"Field not found: {key!r}")
|
|
64
|
+
return found
|
|
65
|
+
|
|
66
|
+
def find(self, key: str) -> Field | None:
|
|
67
|
+
key = str(key).strip()
|
|
68
|
+
lowered = key.lower()
|
|
69
|
+
matchers = (
|
|
70
|
+
lambda f: f.alias == key,
|
|
71
|
+
lambda f: f.ref == key or f.column == key,
|
|
72
|
+
lambda f: f.label.strip().lower() == lowered,
|
|
73
|
+
lambda f: len(key) >= 6 and f.ref.endswith(key),
|
|
74
|
+
)
|
|
75
|
+
for match in matchers:
|
|
76
|
+
hits = [f for f in self.fields if match(f)]
|
|
77
|
+
if len(hits) == 1:
|
|
78
|
+
return hits[0]
|
|
79
|
+
if len(hits) > 1:
|
|
80
|
+
raise KeyError(f"Ambiguous field: {key!r} ({', '.join(f.column for f in hits)})")
|
|
81
|
+
return None
|
|
82
|
+
|
|
83
|
+
def of_kind(self, *kinds) -> list:
|
|
84
|
+
return [f for f in self.fields if f.kind in kinds]
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Converts raw Epicollect responses into types directly used by the UI."""
|
|
2
|
+
import numpy as np
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def _empty_to_nan(value):
|
|
7
|
+
if isinstance(value, str) and value.strip() == "":
|
|
8
|
+
return np.nan
|
|
9
|
+
return value
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _as_list(value):
|
|
13
|
+
if isinstance(value, list):
|
|
14
|
+
return [str(v) for v in value if str(v).strip()]
|
|
15
|
+
if value is None or (isinstance(value, float) and np.isnan(value)):
|
|
16
|
+
return []
|
|
17
|
+
text = str(value).strip()
|
|
18
|
+
# Epicollect CSV brings multiple-choice answers separated by comma
|
|
19
|
+
return [v.strip() for v in text.split(",") if v.strip()] if text else []
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _first(value):
|
|
23
|
+
if isinstance(value, list):
|
|
24
|
+
return value[0] if value else np.nan
|
|
25
|
+
return value
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _coord(value, key):
|
|
29
|
+
if isinstance(value, dict):
|
|
30
|
+
return value.get(key)
|
|
31
|
+
return np.nan
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _unify_case(series):
|
|
35
|
+
"""Typed text responses ("foo bar", "Foo Bar", "FooBar") become the most frequent casing."""
|
|
36
|
+
def key(value):
|
|
37
|
+
return "".join(value.lower().split())
|
|
38
|
+
|
|
39
|
+
counts = series.dropna().value_counts()
|
|
40
|
+
canonical = {}
|
|
41
|
+
for value in counts.index:
|
|
42
|
+
canonical.setdefault(key(value), value)
|
|
43
|
+
return series.map(lambda v: v if pd.isna(v) else canonical[key(v)])
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _categorical(series, options):
|
|
47
|
+
observed = [v for v in series.dropna().unique() if v not in options]
|
|
48
|
+
return pd.Categorical(series, categories=list(options) + sorted(map(str, observed)))
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def normalize(df: pd.DataFrame, fields: list, timezone: str | None = None) -> pd.DataFrame:
|
|
52
|
+
df = df.copy()
|
|
53
|
+
for f in fields:
|
|
54
|
+
if f.column not in df.columns:
|
|
55
|
+
continue
|
|
56
|
+
col = df[f.column]
|
|
57
|
+
kind = f.kind
|
|
58
|
+
|
|
59
|
+
if kind == "location":
|
|
60
|
+
df[f.lat] = pd.to_numeric(col.map(lambda v: _coord(v, "latitude")), errors="coerce")
|
|
61
|
+
df[f.lon] = pd.to_numeric(col.map(lambda v: _coord(v, "longitude")), errors="coerce")
|
|
62
|
+
continue
|
|
63
|
+
|
|
64
|
+
if kind == "multi":
|
|
65
|
+
df[f.column] = col.map(_as_list)
|
|
66
|
+
continue
|
|
67
|
+
|
|
68
|
+
col = col.map(_empty_to_nan)
|
|
69
|
+
|
|
70
|
+
if kind == "categorical":
|
|
71
|
+
col = col.map(_first).map(lambda v: v if pd.isna(v) else " ".join(str(v).split()))
|
|
72
|
+
if not f.options:
|
|
73
|
+
col = _unify_case(col)
|
|
74
|
+
options = f.options or sorted(col.dropna().unique(), key=str.lower)
|
|
75
|
+
df[f.column] = _categorical(col, options)
|
|
76
|
+
elif kind == "numeric":
|
|
77
|
+
df[f.column] = pd.to_numeric(col, errors="coerce")
|
|
78
|
+
elif kind == "date":
|
|
79
|
+
parsed = pd.to_datetime(col, errors="coerce", utc=True, format="mixed")
|
|
80
|
+
if timezone:
|
|
81
|
+
parsed = parsed.dt.tz_convert(timezone)
|
|
82
|
+
df[f.column] = parsed.dt.tz_localize(None)
|
|
83
|
+
else:
|
|
84
|
+
df[f.column] = col
|
|
85
|
+
return df
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def flat_for_export(df: pd.DataFrame, fields: list) -> pd.DataFrame:
|
|
89
|
+
"""Flat table (joined lists, location split into lat/lon) with question labels as column headers."""
|
|
90
|
+
out = pd.DataFrame(index=df.index)
|
|
91
|
+
for f in fields:
|
|
92
|
+
if f.kind == "location":
|
|
93
|
+
if f.lat in df.columns:
|
|
94
|
+
out[f"{f.label} (lat)"] = df[f.lat]
|
|
95
|
+
out[f"{f.label} (lon)"] = df[f.lon]
|
|
96
|
+
elif f.column in df.columns:
|
|
97
|
+
col = df[f.column]
|
|
98
|
+
if f.kind == "multi":
|
|
99
|
+
col = col.map(lambda v: "; ".join(v) if isinstance(v, list) else v)
|
|
100
|
+
elif f.kind == "categorical":
|
|
101
|
+
col = col.astype(object)
|
|
102
|
+
name = f.label if f.label not in out.columns else f"{f.label} [{f.column}]"
|
|
103
|
+
out[name] = col
|
|
104
|
+
return out
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
PAGE_REGISTRY = {}
|
|
2
|
+
SOURCE_REGISTRY = {}
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def page(name: str, order: int = 100, available=None):
|
|
6
|
+
"""Registers a dashboard page. The decorated function receives a PageContext.
|
|
7
|
+
|
|
8
|
+
`available(ctx) -> bool` hides the page when it does not apply to the project.
|
|
9
|
+
"""
|
|
10
|
+
def decorator(func):
|
|
11
|
+
func.page_order = order
|
|
12
|
+
func.page_available = available or (lambda ctx: True)
|
|
13
|
+
PAGE_REGISTRY[name] = func
|
|
14
|
+
return func
|
|
15
|
+
return decorator
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def source(kind: str):
|
|
19
|
+
"""Registers a data source class by the value of `source.type` in config."""
|
|
20
|
+
def decorator(cls):
|
|
21
|
+
SOURCE_REGISTRY[kind] = cls
|
|
22
|
+
return cls
|
|
23
|
+
return decorator
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def ordered_pages() -> dict:
|
|
27
|
+
return dict(sorted(PAGE_REGISTRY.items(), key=lambda item: item[1].page_order))
|