tacit-qda 1.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tacit_qda-1.1.1/LICENSE.txt +21 -0
- tacit_qda-1.1.1/PKG-INFO +561 -0
- tacit_qda-1.1.1/README.md +531 -0
- tacit_qda-1.1.1/frameworks/esg_disclosure_probe.json +226 -0
- tacit_qda-1.1.1/frameworks/ri_stilgoe_2013.json +332 -0
- tacit_qda-1.1.1/frameworks/tdf_cane_2012.json +594 -0
- tacit_qda-1.1.1/frameworks/utaut_venkatesh_2003.json +226 -0
- tacit_qda-1.1.1/pyproject.toml +54 -0
- tacit_qda-1.1.1/setup.cfg +4 -0
- tacit_qda-1.1.1/src/__init__.py +53 -0
- tacit_qda-1.1.1/src/_demo_en_academia.py +639 -0
- tacit_qda-1.1.1/src/_demo_en_government.py +623 -0
- tacit_qda-1.1.1/src/_demo_en_industry.py +636 -0
- tacit_qda-1.1.1/src/_demo_en_nonprofit.py +622 -0
- tacit_qda-1.1.1/src/_launch_common.py +348 -0
- tacit_qda-1.1.1/src/_tacit_cli.py +28 -0
- tacit_qda-1.1.1/src/app.py +3781 -0
- tacit_qda-1.1.1/src/bench_agreement.py +134 -0
- tacit_qda-1.1.1/src/bench_models.py +481 -0
- tacit_qda-1.1.1/src/bench_open_coding.py +329 -0
- tacit_qda-1.1.1/src/bench_open_summary.py +181 -0
- tacit_qda-1.1.1/src/bench_yield.py +140 -0
- tacit_qda-1.1.1/src/demo_transcripts_en.py +238 -0
- tacit_qda-1.1.1/src/demo_transcripts_zh.py +333 -0
- tacit_qda-1.1.1/src/make_demo_data.py +427 -0
- tacit_qda-1.1.1/src/make_hearing_corpus.py +547 -0
- tacit_qda-1.1.1/src/tacit_analysis.py +428 -0
- tacit_qda-1.1.1/src/tacit_coding.py +321 -0
- tacit_qda-1.1.1/src/tacit_framework.py +977 -0
- tacit_qda-1.1.1/src/tacit_i18n.py +496 -0
- tacit_qda-1.1.1/src/tacit_irr.py +586 -0
- tacit_qda-1.1.1/src/tacit_lexicon.py +1125 -0
- tacit_qda-1.1.1/src/tacit_llm.py +1029 -0
- tacit_qda-1.1.1/src/tacit_open.py +755 -0
- tacit_qda-1.1.1/src/tacit_openalex.py +1184 -0
- tacit_qda-1.1.1/src/tacit_review.py +384 -0
- tacit_qda-1.1.1/src/tacit_schema.py +876 -0
- tacit_qda-1.1.1/src/tacit_strings.py +1828 -0
- tacit_qda-1.1.1/src/tacit_themes.py +1056 -0
- tacit_qda-1.1.1/tacit_qda.egg-info/PKG-INFO +561 -0
- tacit_qda-1.1.1/tacit_qda.egg-info/SOURCES.txt +63 -0
- tacit_qda-1.1.1/tacit_qda.egg-info/dependency_links.txt +1 -0
- tacit_qda-1.1.1/tacit_qda.egg-info/entry_points.txt +2 -0
- tacit_qda-1.1.1/tacit_qda.egg-info/requires.txt +11 -0
- tacit_qda-1.1.1/tacit_qda.egg-info/top_level.txt +1 -0
- tacit_qda-1.1.1/tests/test_analysis_engine.py +320 -0
- tacit_qda-1.1.1/tests/test_app_functions.py +405 -0
- tacit_qda-1.1.1/tests/test_bench_resume.py +146 -0
- tacit_qda-1.1.1/tests/test_coding.py +310 -0
- tacit_qda-1.1.1/tests/test_demo_data.py +206 -0
- tacit_qda-1.1.1/tests/test_framework.py +439 -0
- tacit_qda-1.1.1/tests/test_frameworks_shipped.py +293 -0
- tacit_qda-1.1.1/tests/test_i18n.py +164 -0
- tacit_qda-1.1.1/tests/test_irr.py +406 -0
- tacit_qda-1.1.1/tests/test_launcher.py +339 -0
- tacit_qda-1.1.1/tests/test_lexicon.py +436 -0
- tacit_qda-1.1.1/tests/test_llm_providers.py +961 -0
- tacit_qda-1.1.1/tests/test_llm_retry.py +283 -0
- tacit_qda-1.1.1/tests/test_open_coding.py +568 -0
- tacit_qda-1.1.1/tests/test_openalex.py +635 -0
- tacit_qda-1.1.1/tests/test_review.py +253 -0
- tacit_qda-1.1.1/tests/test_schema.py +451 -0
- tacit_qda-1.1.1/tests/test_themes.py +678 -0
- tacit_qda-1.1.1/tests/test_ui_full.py +394 -0
- tacit_qda-1.1.1/tests/test_ui_smoke.py +915 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Hung-Chi Chang
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
tacit_qda-1.1.1/PKG-INFO
ADDED
|
@@ -0,0 +1,561 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tacit-qda
|
|
3
|
+
Version: 1.1.1
|
|
4
|
+
Summary: Auditable, locally deployable LLM-assisted qualitative coding with codebook provenance
|
|
5
|
+
Author: Hung-Chi Chang
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/NYCU-FUN-TECH-LAB/TACIT
|
|
8
|
+
Project-URL: Documentation, https://github.com/NYCU-FUN-TECH-LAB/TACIT/blob/main/README.md
|
|
9
|
+
Project-URL: Archive, https://doi.org/10.5281/zenodo.22882168
|
|
10
|
+
Keywords: qualitative research,qualitative coding,thematic analysis,large language models,codebook,inter-rater reliability,CAQDAS
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
15
|
+
Classifier: Topic :: Sociology
|
|
16
|
+
Requires-Python: >=3.9
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE.txt
|
|
19
|
+
Requires-Dist: streamlit>=1.31
|
|
20
|
+
Requires-Dist: python-docx>=1.1
|
|
21
|
+
Requires-Dist: pandas>=2.0
|
|
22
|
+
Requires-Dist: numpy>=1.24
|
|
23
|
+
Requires-Dist: openpyxl>=3.1
|
|
24
|
+
Requires-Dist: google-genai>=1.0
|
|
25
|
+
Requires-Dist: plotly>=5.18
|
|
26
|
+
Requires-Dist: scipy>=1.11
|
|
27
|
+
Provides-Extra: zh
|
|
28
|
+
Requires-Dist: ckip-transformers; extra == "zh"
|
|
29
|
+
Dynamic: license-file
|
|
30
|
+
|
|
31
|
+
# TACIT
|
|
32
|
+
|
|
33
|
+
**Theory-Anchored Coding with Interpretive Transparency.**
|
|
34
|
+
|
|
35
|
+
A Streamlit application that codes interview transcripts against a *pluggable*
|
|
36
|
+
theoretical framework, keeps interpretive authority with the researcher, and
|
|
37
|
+
reports the statistics a reader will actually ask for — including inter-rater
|
|
38
|
+
reliability computed from genuine double-blind double coding.
|
|
39
|
+
|
|
40
|
+
Language models can run **entirely on your own machine**, so transcripts need
|
|
41
|
+
not leave it.
|
|
42
|
+
|
|
43
|
+
[](LICENSE.txt)
|
|
44
|
+
[](https://www.python.org/)
|
|
45
|
+
[](#testing)
|
|
46
|
+
[](https://doi.org/10.5281/zenodo.22882168)
|
|
47
|
+
|
|
48
|
+
---
|
|
49
|
+
|
|
50
|
+
## What makes this different
|
|
51
|
+
|
|
52
|
+
Most qualitative tools treat a language model as an autocoder: feed it a
|
|
53
|
+
transcript, get codes back, accept them. This one refuses to work that way, for
|
|
54
|
+
four reasons.
|
|
55
|
+
|
|
56
|
+
**The framework comes from literature, not from the model's memory.**
|
|
57
|
+
A coding framework is the validity foundation of the whole study. Asking a model
|
|
58
|
+
to recall the dimensions of a theory produces output that is plausible and
|
|
59
|
+
partly invented, including citations to work that does not exist. Here you
|
|
60
|
+
retrieve literature from OpenAlex, the model drafts dimensions *from the
|
|
61
|
+
retrieved abstracts*, and a hallucination guard rejects any citation that was
|
|
62
|
+
not actually retrieved. You then approve the framework dimension by dimension,
|
|
63
|
+
under your own name.
|
|
64
|
+
|
|
65
|
+
**Interpretive authority stays with the researcher.**
|
|
66
|
+
Model coding is a draft. Every code can be confirmed, revised or deleted, and
|
|
67
|
+
you can add segments the model missed. Every change is written to an audit
|
|
68
|
+
trail, and the model's original judgement is retained — so "how often did the
|
|
69
|
+
model need correcting?" becomes a reportable number rather than an unknown.
|
|
70
|
+
|
|
71
|
+
**Anything computable is computed, not estimated.**
|
|
72
|
+
Reliability uses a sampling frame that *includes unmarked units*, so recall is
|
|
73
|
+
meaningful — without them, recall is identically 1 and the number is
|
|
74
|
+
meaningless. When the assumptions of a chi-square test fail, the tool refuses to
|
|
75
|
+
print a p-value rather than inviting a claim that cannot survive scrutiny.
|
|
76
|
+
|
|
77
|
+
**Your data does not have to leave your machine.**
|
|
78
|
+
Interview transcripts are human-subject data. Whether they may be sent to a
|
|
79
|
+
third-party service is decided by your ethics approval, and approvals written
|
|
80
|
+
before generative AI generally do not cover it. Point TACIT at a local model
|
|
81
|
+
server and nothing goes over the network. It also helps reproducibility: cloud
|
|
82
|
+
models get withdrawn, and the weights behind a version string change silently,
|
|
83
|
+
whereas a local model has a fixed weight file you can name in a paper.
|
|
84
|
+
|
|
85
|
+
---
|
|
86
|
+
|
|
87
|
+
## Try it
|
|
88
|
+
|
|
89
|
+
| | |
|
|
90
|
+
|---|---|
|
|
91
|
+
| **Manual (English)** | `docs/TACIT_User-Manual_en.pdf` |
|
|
92
|
+
| **Manual (中文)** | `docs/TACIT_User-Manual_zh.pdf` |
|
|
93
|
+
| **Demo corpora** | 80 witness transcripts from U.S. congressional hearings on AI governance (public domain) · 24 synthetic English interviews with reference codings · 6 Traditional Chinese |
|
|
94
|
+
|
|
95
|
+
The synthetic interviews ship with **a reference coding for every one of
|
|
96
|
+
them**, so you can open every tab and see real results with **no API key and
|
|
97
|
+
no local model**. The hearing transcripts are real speech you can check against
|
|
98
|
+
its source. See [Demonstration data](#demonstration-data).
|
|
99
|
+
|
|
100
|
+
---
|
|
101
|
+
|
|
102
|
+
## Evaluating TACIT without a model
|
|
103
|
+
|
|
104
|
+
You do not need a model to evaluate this software. There are three levels, and
|
|
105
|
+
the first two need nothing installed beyond the requirements.
|
|
106
|
+
|
|
107
|
+
**Level 1 — no model, no key, our data.** Install, launch, and in the sidebar
|
|
108
|
+
tick the shipped analyses under `analyses/`. Every analysis tab then works on
|
|
109
|
+
real coded material: cross-tabulations, co-occurrence, the cross-case matrix,
|
|
110
|
+
the audit trail in the coding-review tab, the reliability report, and the
|
|
111
|
+
Excel/Gioia exports. This exercises everything except the calls to the model
|
|
112
|
+
itself. The shipped codings are read-only — editing one saves a new file and
|
|
113
|
+
records what it was derived from, so the reference material cannot be
|
|
114
|
+
overwritten by accident.
|
|
115
|
+
|
|
116
|
+
**Level 2 — no model, no key, your own transcripts.** Open the reliability tab
|
|
117
|
+
and set *Build the frame from* to *Transcripts I upload here*. Upload your own
|
|
118
|
+
`.docx` files; TACIT splits them into speaking units, draws a blinded sample
|
|
119
|
+
from a seed you choose, and issues one coding sheet per coder. Take the
|
|
120
|
+
completed sheets back and you get percentage agreement, Cohen's κ, PABAK,
|
|
121
|
+
Gwet's AC1 and Krippendorff's α, per code and pooled, with the disagreement
|
|
122
|
+
list.
|
|
123
|
+
|
|
124
|
+
That is not the end of it. *Turn these codings into an analysis* sends a
|
|
125
|
+
coder's completed sheet into the workspace as ordinary records, and every
|
|
126
|
+
analysis tab then works on them: crosstabs by case descriptor with the
|
|
127
|
+
chi-square guard, co-occurrence, the cross-case matrix, lexicon discovery, the
|
|
128
|
+
Excel and Gioia exports — and theme grouping, which clusters the codes by
|
|
129
|
+
co-occurrence or by label similarity and leaves the naming to you. A complete
|
|
130
|
+
two-coder thematic analysis, from transcript to data structure, with no model
|
|
131
|
+
involved at any point and nothing leaving the machine.
|
|
132
|
+
|
|
133
|
+
**Level 3 — with a model.** Either a free Google Gemini key
|
|
134
|
+
(<https://aistudio.google.com/apikey>, no payment details required) or a local
|
|
135
|
+
Ollama model. See [Choosing a model provider](#choosing-a-model-provider).
|
|
136
|
+
|
|
137
|
+
One thing about the free Gemini tier is worth knowing before you plan a run,
|
|
138
|
+
because it is easy to misread as a broken key. **The daily request quota is per
|
|
139
|
+
project and per model, not per key**, so issuing a second key in the same
|
|
140
|
+
project changes nothing. The limit differs sharply between models: as of
|
|
141
|
+
September 2026 `gemini-3.6-flash` allowed 20 requests per day, which will not
|
|
142
|
+
finish a single transcript, while `gemini-3.5-flash-lite` completed a
|
|
143
|
+
12-transcript windowed run (149 requests) with room to spare. A 429 reply names
|
|
144
|
+
the quota it hit — look for `GenerateRequestsPerDayPerProjectPerModel` and the
|
|
145
|
+
`limit:` value — and TACIT reports quota errors separately from a busy server,
|
|
146
|
+
because waiting helps with the second and not with the first.
|
|
147
|
+
|
|
148
|
+
---
|
|
149
|
+
|
|
150
|
+
## Install
|
|
151
|
+
|
|
152
|
+
Requires Python 3.9+.
|
|
153
|
+
|
|
154
|
+
```bash
|
|
155
|
+
git clone https://github.com/NYCU-FUN-TECH-LAB/TACIT.git
|
|
156
|
+
cd TACIT
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
Then double-click the launcher for your platform:
|
|
160
|
+
|
|
161
|
+
| Platform | Launcher |
|
|
162
|
+
|---|---|
|
|
163
|
+
| Windows | `launch_windows.bat` |
|
|
164
|
+
| macOS | `launch_mac.command` |
|
|
165
|
+
| Linux | `launch_linux.sh` |
|
|
166
|
+
|
|
167
|
+
The launcher creates an isolated environment, installs dependencies, picks a
|
|
168
|
+
free port and opens a browser. First run takes 3–10 minutes.
|
|
169
|
+
|
|
170
|
+
Prefer to do it manually:
|
|
171
|
+
|
|
172
|
+
```bash
|
|
173
|
+
python -m venv .venv && source .venv/bin/activate # Windows: .venv\Scripts\activate
|
|
174
|
+
pip install -r requirements.txt
|
|
175
|
+
streamlit run src/app.py
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
### Install from PyPI
|
|
179
|
+
|
|
180
|
+
TACIT is also published as a Python package:
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
pip install tacit-qda
|
|
184
|
+
tacit-qda
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
`tacit-qda` starts the interface with the current directory as the workspace:
|
|
188
|
+
frameworks, analyses and lexicons are read from and written there, and the four
|
|
189
|
+
shipped frameworks are copied into `frameworks/` on first use. Arguments after
|
|
190
|
+
the command go to `streamlit run` (for example `tacit-qda --server.port 8600`).
|
|
191
|
+
The demonstration corpora and benchmark scripts are in this repository, not in
|
|
192
|
+
the package. `pip install "tacit-qda[zh]"` adds the Traditional Chinese segmenter.
|
|
193
|
+
|
|
194
|
+
The engines can also be used from a script:
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
import tacit_qda # puts the TACIT modules on the import path
|
|
198
|
+
import tacit_framework as F
|
|
199
|
+
F.FRAMEWORK_DIR = tacit_qda.bundled_frameworks_dir()
|
|
200
|
+
F.activate_by_id("ri_stilgoe_2013")
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
---
|
|
204
|
+
|
|
205
|
+
## Choosing a model provider
|
|
206
|
+
|
|
207
|
+
Pick the provider at the top of the sidebar. Everything downstream — coding,
|
|
208
|
+
theme induction, framework drafting — uses whichever you chose.
|
|
209
|
+
|
|
210
|
+
| Provider | Needs a key | Data leaves your machine | Extra Python package |
|
|
211
|
+
|---|---|---|---|
|
|
212
|
+
| **Ollama** | no | **no** | none |
|
|
213
|
+
| **OpenAI-compatible endpoint** (LM Studio, llama.cpp server, vLLM, LocalAI) | no, if it is on localhost | **no**, if it is on localhost | none |
|
|
214
|
+
| **Google Gemini** | yes | yes | `google-genai` (installed with the requirements) |
|
|
215
|
+
|
|
216
|
+
All three providers work straight after installation. The Google SDK is part
|
|
217
|
+
of `requirements.txt`, so a machine without Ollama or any other local model
|
|
218
|
+
server can use Gemini as soon as TACIT is installed. The local providers use
|
|
219
|
+
only the Python standard library.
|
|
220
|
+
|
|
221
|
+
The older `google-generativeai` package still works and is detected
|
|
222
|
+
automatically, but Google has ended support for it; the sidebar will say so and
|
|
223
|
+
point at the replacement.
|
|
224
|
+
|
|
225
|
+
### Local setup in three commands
|
|
226
|
+
|
|
227
|
+
```bash
|
|
228
|
+
ollama serve
|
|
229
|
+
ollama pull llama3.1:8b-instruct-q4_K_M
|
|
230
|
+
streamlit run src/app.py
|
|
231
|
+
```
|
|
232
|
+
|
|
233
|
+
Then choose **Ollama (local)** in the sidebar, set the address to
|
|
234
|
+
`http://localhost:11434`, and press **Test connection**.
|
|
235
|
+
|
|
236
|
+
### One thing you must get right: the context window
|
|
237
|
+
|
|
238
|
+
Ollama's default context window is 4,096 tokens. **One interview does not fit.**
|
|
239
|
+
A server that runs out of context truncates the input *silently* and still
|
|
240
|
+
returns well-formed JSON — so your coding looks complete while half the
|
|
241
|
+
transcript was never read. This is the most dangerous failure mode in local
|
|
242
|
+
deployment precisely because nothing goes wrong on screen.
|
|
243
|
+
|
|
244
|
+
TACIT therefore sets the context window explicitly (default 32,768), exposes it
|
|
245
|
+
in the sidebar, estimates the prompt length before sending, and **refuses to
|
|
246
|
+
send a prompt that would not fit**. Raise it for long transcripts; it costs
|
|
247
|
+
memory, roughly 0.5–2 GB per 32k tokens depending on the model.
|
|
248
|
+
|
|
249
|
+
### Batch and scripted runs
|
|
250
|
+
|
|
251
|
+
```bash
|
|
252
|
+
export TACIT_PROVIDER=ollama
|
|
253
|
+
export TACIT_MODEL=llama3.1:8b-instruct-q4_K_M
|
|
254
|
+
export TACIT_BASE_URL=http://localhost:11434
|
|
255
|
+
export TACIT_NUM_CTX=32768
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
The sidebar picks these up as defaults.
|
|
259
|
+
|
|
260
|
+
### What gets recorded
|
|
261
|
+
|
|
262
|
+
Every analysis record stores the **full endpoint descriptor**, not just the
|
|
263
|
+
model name — `ollama/llama3.1:8b-instruct-q4_K_M@http://localhost:11434` rather
|
|
264
|
+
than `llama3.1:8b`. The same short name means different quantisations on
|
|
265
|
+
different machines, and a record that only names the model cannot be
|
|
266
|
+
reproduced.
|
|
267
|
+
|
|
268
|
+
---
|
|
269
|
+
|
|
270
|
+
## Pluggable frameworks
|
|
271
|
+
|
|
272
|
+
Three frameworks ship with the tool, and they are deliberately unalike — each
|
|
273
|
+
one exercises a different part of the framework contract:
|
|
274
|
+
|
|
275
|
+
| Framework | Dimensions | Polarity | Field | What it tests |
|
|
276
|
+
|---|---|---|---|---|
|
|
277
|
+
| `ri_stilgoe_2013` | Anticipation, Reflexivity, Engagement, Responsiveness | yes, `P`/`N` → 8 codes | Responsible innovation | the default |
|
|
278
|
+
| `utaut_venkatesh_2003` | Performance expectancy, Effort expectancy, Social influence, Facilitating conditions | none → 4 codes | Technology acceptance | a framework with **no polarity model** |
|
|
279
|
+
| `esg_disclosure_probe` | Targets and baselines, Measurement and boundary, Governance and accountability, Stakeholder engagement | yes, `S`/`A` → 8 codes | Sustainability disclosure | polarity values that are **not** `P`/`N`; a non-interview corpus |
|
|
280
|
+
|
|
281
|
+
None of them is hard-coded. Dimensions, polarity values, descriptor fields and
|
|
282
|
+
interface labels live in a JSON file, and the analysis engines, prompts and
|
|
283
|
+
schema all derive from whichever framework is active. Switching to UTAUT
|
|
284
|
+
collapses codes from `ANT-P`/`ANT-N` to `PE`, and the polarity sub-tab
|
|
285
|
+
disappears because that analysis is meaningless without a polarity model.
|
|
286
|
+
|
|
287
|
+
The third framework guards against a specific failure: code that derives the *dimensions* from the active framework but still wrote the two
|
|
288
|
+
polarity values as the literals `"P"` and `"N"` in the polarity-balance
|
|
289
|
+
statistics and in the lexicon coder — which no test caught, because the only
|
|
290
|
+
other framework that shipped had no polarity model at all and skipped those code
|
|
291
|
+
paths entirely. `esg_disclosure_probe` names its poles `S` (substantiated) and
|
|
292
|
+
`A` (asserted), so it does exercise them. `tests/test_frameworks_shipped.py`
|
|
293
|
+
runs every framework in `frameworks/` through the full analysis pipeline for
|
|
294
|
+
exactly this reason: a pluggability claim is only worth as much as the least
|
|
295
|
+
similar framework you have actually run.
|
|
296
|
+
|
|
297
|
+
Any deductive coding scheme is in scope — template analysis, framework analysis,
|
|
298
|
+
theory-driven content analysis. Write a JSON file, or build one from retrieved
|
|
299
|
+
literature in the Framework builder tab.
|
|
300
|
+
|
|
301
|
+
---
|
|
302
|
+
|
|
303
|
+
## Demonstration data
|
|
304
|
+
|
|
305
|
+
Two corpora ship, and they are different kinds of thing. Do not mix them up.
|
|
306
|
+
|
|
307
|
+
| | Congressional hearings | Synthetic interviews | Synthetic, Chinese |
|
|
308
|
+
|---|---|---|---|
|
|
309
|
+
| Folder | `demo_data/hearings/` | `demo_data/en/` | `demo_data/zh/` |
|
|
310
|
+
| What it is | **real** testimony, public record | **fictional**, written by a language model | fictional |
|
|
311
|
+
| Transcripts | 80 witnesses, 22 hearings (24 form the paper's subset) | 24 | 6 |
|
|
312
|
+
| Sectors | industry, academia, government, civil society | the same four, six each | 4 |
|
|
313
|
+
| Reference codings | none | 24, in `analyses/` | — |
|
|
314
|
+
| Use it for | open coding, screenshots, anything you want a reader to be able to verify | statistics, agreement against a fixed standard, regression tests | CKIP segmentation, language routing |
|
|
315
|
+
|
|
316
|
+
**The hearings** are U.S. congressional hearings on artificial-intelligence
|
|
317
|
+
governance, March 2023 to June 2025, retrieved from
|
|
318
|
+
[govinfo.gov](https://www.govinfo.gov) and in the public domain under
|
|
319
|
+
17 U.S.C. § 105. The record was split at witness level, one file per speaker;
|
|
320
|
+
prepared statements reprinted in the record were left out because they are
|
|
321
|
+
written documents, not testimony. `manifest.json` gives the package identifier,
|
|
322
|
+
hearing title, date, chamber and source URL for every witness, the hand-assigned
|
|
323
|
+
sector, and the three witnesses excluded with the reason for each.
|
|
324
|
+
|
|
325
|
+
```bash
|
|
326
|
+
python src/make_hearing_corpus.py # re-fetches from govinfo and rebuilds the corpus
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
**The synthetic interviews** are fictional. Transcripts and reference codings
|
|
330
|
+
alike were written by a language model (Claude) against the responsible-innovation
|
|
331
|
+
codebook, so every quotation is by construction a clean substring of a passage
|
|
332
|
+
that exemplifies its code. That is what makes them a fixed comparison standard,
|
|
333
|
+
and it is also why agreement measured against them is an **upper bound** —
|
|
334
|
+
expect lower figures on real transcripts. They must not be cited as empirical
|
|
335
|
+
data. Read `demo_data/en/00_ABOUT_THIS_DATA.txt` before using them.
|
|
336
|
+
|
|
337
|
+
```bash
|
|
338
|
+
python src/make_demo_data.py # rebuilds the .docx files and reference codings
|
|
339
|
+
```
|
|
340
|
+
|
|
341
|
+
### Why twenty-four and not six
|
|
342
|
+
|
|
343
|
+
Six transcripts are enough to walk through the interface and not enough for a
|
|
344
|
+
single inferential statistic to run. In a 2×4 crosstab of institution type
|
|
345
|
+
against polarity, expected cell counts land around one or two, the chi-square
|
|
346
|
+
assumptions fail, and the tool correctly refuses to print a p-value. A newcomer
|
|
347
|
+
then opens the cross-analysis tab and sees a column of insufficient-data
|
|
348
|
+
notices — the part of the tool most worth examining shows nothing at all.
|
|
349
|
+
|
|
350
|
+
With 24 respondents the corpus produces:
|
|
351
|
+
|
|
352
|
+
| Quantity | Value |
|
|
353
|
+
|---|---|
|
|
354
|
+
| Coded segments | 148, of which 36 (24%) are coded more than once |
|
|
355
|
+
| Codes assigned | 184 (one per researcher judgement) |
|
|
356
|
+
| Analysis units | 181 distinct segment × dimension × polarity |
|
|
357
|
+
| Reliability sampling frame | 299 units, **174 of them unmarked** |
|
|
358
|
+
| Institution type × polarity | χ²(3) = 10.04, *p* = .018, *V* = .236, min. expected 18.1 |
|
|
359
|
+
| Institution type × dimension | χ²(9) = 7.19, *p* = .617, *V* = .115, assumptions met |
|
|
360
|
+
| Institution type × full code set | assumptions not met, p value withheld |
|
|
361
|
+
|
|
362
|
+
Those three rows are all deliberate: one significant, one null with assumptions
|
|
363
|
+
met, one withheld. A demonstration corpus in which every test came out
|
|
364
|
+
significant would be demonstrating the corpus, not the method.
|
|
365
|
+
|
|
366
|
+
The two code counts differ because they answer different questions. Three
|
|
367
|
+
segments carry two judgements on the same dimension and polarity, with
|
|
368
|
+
different rationales; the record keeps both, because a rationale is the
|
|
369
|
+
researcher's reasoning and not an annotation. The analysis layer counts each
|
|
370
|
+
segment once per dimension and polarity, because counting one passage twice
|
|
371
|
+
under the same code would inflate its co-occurrence and Jaccard weights.
|
|
372
|
+
Both numbers are correct; any report has to say which one it is using.
|
|
373
|
+
|
|
374
|
+
### About the reference codings
|
|
375
|
+
|
|
376
|
+
The codings in `analyses/` were **written alongside the transcripts**. They are
|
|
377
|
+
not the output of any particular model run and should not be read as evidence of
|
|
378
|
+
how well any model performs. They exist so every tab can be opened without an
|
|
379
|
+
API key.
|
|
380
|
+
|
|
381
|
+
Using them as a comparison set for reliability *is* a legitimate use: run the
|
|
382
|
+
coder yourself, then compare your run against this fixed, deliberately designed
|
|
383
|
+
scheme.
|
|
384
|
+
|
|
385
|
+
---
|
|
386
|
+
|
|
387
|
+
## Handling real interview data
|
|
388
|
+
|
|
389
|
+
**Keep confidential transcripts outside this folder.** `.gitignore` only
|
|
390
|
+
excludes files that are not already tracked; it is a safety net, not a
|
|
391
|
+
guarantee. A single `git add -f` defeats it.
|
|
392
|
+
|
|
393
|
+
```
|
|
394
|
+
project/
|
|
395
|
+
├── tacit/ ← this repository
|
|
396
|
+
└── research-data/ ← your transcripts and analyses, never committed
|
|
397
|
+
```
|
|
398
|
+
|
|
399
|
+
If your ethics approval does not permit sending transcripts to a third-party
|
|
400
|
+
service, use a local provider and verify the address is on `localhost`. The
|
|
401
|
+
sidebar states which case you are in.
|
|
402
|
+
|
|
403
|
+
---
|
|
404
|
+
|
|
405
|
+
## What's in each tab
|
|
406
|
+
|
|
407
|
+
| Tab | Purpose |
|
|
408
|
+
|---|---|
|
|
409
|
+
| Run analysis | Code transcripts against the active framework. The only step that uses model capacity |
|
|
410
|
+
| Data & descriptors | Edit respondent attributes; these drive every later comparison |
|
|
411
|
+
| Code review | Confirm, revise, delete or add codes. Full audit trail |
|
|
412
|
+
| Cross-analysis | Crosstabs, co-occurrence, cross-case matrix, polarity balance |
|
|
413
|
+
| Theme structure | Two-stage inductive theme induction with a model, or model-free grouping of the codes by co-occurrence or label similarity which you name yourself; Gioia data structure figure with SVG export |
|
|
414
|
+
| Lexicon induction | Term discovery, log-odds feature induction, dictionary baseline for auditing the model |
|
|
415
|
+
| Reliability | Sampling frame from records or from transcripts you upload, double-blind coding sheets, κ / PABAK / AC1 / Krippendorff α, confusion matrices, model precision and recall, and human codings turned back into analysable records |
|
|
416
|
+
| Export | Excel, Word, JSON |
|
|
417
|
+
| Framework builder | Retrieve literature from OpenAlex, draft dimensions, approve them individually |
|
|
418
|
+
|
|
419
|
+
---
|
|
420
|
+
|
|
421
|
+
## Reproducing the numbers in the paper
|
|
422
|
+
|
|
423
|
+
Every figure quoted in the SoftwareX article comes from one of the scripts
|
|
424
|
+
below, and each writes its raw output under `bench_out/`, which is kept in this
|
|
425
|
+
repository. Run them from the project root with the virtual environment active.
|
|
426
|
+
|
|
427
|
+
| What it produces | Command |
|
|
428
|
+
|---|---|
|
|
429
|
+
| Demonstration corpus and reference codings (Table 7) | `python src/make_demo_data.py` then `python run_tests.py demo` |
|
|
430
|
+
| Model comparison: segments, codes, κ against the reference codings, wall time (Table 6) | `python src/bench_models.py --provider ollama --models llama3:8b` |
|
|
431
|
+
| The same for a cloud model | `python src/bench_models.py --provider gemini --models gemini-3.5-flash-lite --api-key <key>` |
|
|
432
|
+
| Agreement against the reference codings on the 299-unit frame (Section 3) | `python src/bench_agreement.py bench_out/local_8b/records/llama3_8b/en` |
|
|
433
|
+
| Single-pass against windowed yield (Section 2.4) | `python src/bench_yield.py --model llama3:8b` |
|
|
434
|
+
| Open coding, one run on the twelve-transcript subset (Table 8) | `python src/bench_open_coding.py --corpus hearings --model llama3:8b --per-sector 3 --tag 12_run1` |
|
|
435
|
+
| The same with a cloud model | `python src/bench_open_coding.py --corpus hearings --provider gemini --model gemini-3.5-flash-lite --per-sector 3 --tag 12_run1` |
|
|
436
|
+
| Table 8 itself: median and range over the archived runs, no model needed | `python src/bench_open_summary.py` |
|
|
437
|
+
| Figure 2, computed from the shipped context formula | `python paper/make_fig2.py` |
|
|
438
|
+
| Figure 6, the three-level data structure | `python paper/make_fig6.py` |
|
|
439
|
+
|
|
440
|
+
Two caveats about exactness.
|
|
441
|
+
|
|
442
|
+
**The cloud row cannot be reproduced identically.** Weights behind a stable
|
|
443
|
+
version string change and older versions are withdrawn, which is one of the
|
|
444
|
+
arguments the article makes. The archived outputs under
|
|
445
|
+
`bench_out/cloud/` are therefore the record of what that run did; re-running
|
|
446
|
+
gives a comparable but not identical result.
|
|
447
|
+
|
|
448
|
+
**Runs vary between samples, by a lot.** On the same twelve transcripts, in the
|
|
449
|
+
same order, at the same temperature, the cloud model returned 89, 143 and 156
|
|
450
|
+
codes and the local 8B model 11, 27 and 108. One open-coding run is therefore
|
|
451
|
+
not a reportable number. Give each run its own `--tag` (the archived ones are
|
|
452
|
+
`12_run1`, `12_run2`, `12_run3`) so it lands in its own directory; `bench_open_summary.py` then reports the median and range, and
|
|
453
|
+
leaves out runs made by a different version of the code, saying which and why.
|
|
454
|
+
|
|
455
|
+
---
|
|
456
|
+
|
|
457
|
+
## Testing
|
|
458
|
+
|
|
459
|
+
```bash
|
|
460
|
+
python run_tests.py
|
|
461
|
+
```
|
|
462
|
+
|
|
463
|
+
Seventeen suites, **none requiring network access, an API key or a model
|
|
464
|
+
server**. The OpenAlex client is tested against offline fixtures, the provider
|
|
465
|
+
abstraction against fake transports, and the interface end-to-end with
|
|
466
|
+
Streamlit's `AppTest` in both languages and several frameworks.
|
|
467
|
+
|
|
468
|
+
`tests/test_frameworks_shipped.py` enumerates the real `frameworks/` directory
|
|
469
|
+
rather than a hard-coded list, and pushes every framework it finds through the
|
|
470
|
+
whole pipeline — schema, prompt, cross-tabs, co-occurrence, polarity balance,
|
|
471
|
+
lexicon, review, themes, reliability sampling and interface labels. Adding a
|
|
472
|
+
framework file adds it to the test run automatically. This suite exists because
|
|
473
|
+
the pluggability claim had been tested only against frameworks that happened to
|
|
474
|
+
resemble the default one.
|
|
475
|
+
|
|
476
|
+
Run one suite:
|
|
477
|
+
|
|
478
|
+
```bash
|
|
479
|
+
python run_tests.py demo
|
|
480
|
+
```
|
|
481
|
+
|
|
482
|
+
---
|
|
483
|
+
|
|
484
|
+
## Optional: Traditional Chinese segmentation
|
|
485
|
+
|
|
486
|
+
```bash
|
|
487
|
+
pip install ckip-transformers
|
|
488
|
+
```
|
|
489
|
+
|
|
490
|
+
Installed automatically by the launchers. Without it the lexicon module falls
|
|
491
|
+
back to an n-gram heuristic with no linguistic knowledge of Chinese, and some
|
|
492
|
+
entries will be wrong word boundaries. Semantic coding does not pass through
|
|
493
|
+
segmentation and is unaffected.
|
|
494
|
+
|
|
495
|
+
---
|
|
496
|
+
|
|
497
|
+
## Citation
|
|
498
|
+
|
|
499
|
+
If this tool contributes to published work, please cite it. See
|
|
500
|
+
[CITATION.cff](CITATION.cff); machine-readable metadata is in
|
|
501
|
+
[codemeta.json](codemeta.json).
|
|
502
|
+
|
|
503
|
+
Every release is archived on Zenodo. Version 1.0.0 is
|
|
504
|
+
[10.5281/zenodo.22882169](https://doi.org/10.5281/zenodo.22882169);
|
|
505
|
+
[10.5281/zenodo.22882168](https://doi.org/10.5281/zenodo.22882168) always resolves to
|
|
506
|
+
the latest version.
|
|
507
|
+
|
|
508
|
+
The built-in frameworks rest on:
|
|
509
|
+
|
|
510
|
+
- Stilgoe, J., Owen, R., & Macnaghten, P. (2013). Developing a framework for
|
|
511
|
+
responsible innovation. *Research Policy, 42*(9), 1568–1580.
|
|
512
|
+
https://doi.org/10.1016/j.respol.2013.05.008
|
|
513
|
+
- Venkatesh, V., Morris, M. G., Davis, G. B., & Davis, F. D. (2003). User
|
|
514
|
+
acceptance of information technology: Toward a unified view. *MIS Quarterly,
|
|
515
|
+
27*(3), 425–478. https://doi.org/10.2307/30036540
|
|
516
|
+
|
|
517
|
+
Statistical methods:
|
|
518
|
+
|
|
519
|
+
- Monroe, B. L., Colaresi, M. P., & Quinn, K. M. (2008). Fightin' words.
|
|
520
|
+
*Political Analysis, 16*(4), 372–403.
|
|
521
|
+
- Gwet, K. L. (2008). Computing inter-rater reliability and its variance in the
|
|
522
|
+
presence of high agreement. *British Journal of Mathematical and Statistical
|
|
523
|
+
Psychology, 61*(1), 29–48.
|
|
524
|
+
- Gioia, D. A., Corley, K. G., & Hamilton, A. L. (2013). Seeking qualitative
|
|
525
|
+
rigor in inductive research. *Organizational Research Methods, 16*(1), 15–31.
|
|
526
|
+
|
|
527
|
+
---
|
|
528
|
+
|
|
529
|
+
## Limitations
|
|
530
|
+
|
|
531
|
+
- Model coding is a draft, not ground truth. Unreviewed coding should not be
|
|
532
|
+
written up as verified.
|
|
533
|
+
- Small local models produce noticeably fewer segments and follow the output
|
|
534
|
+
format less reliably than large ones. If a model keeps failing to return
|
|
535
|
+
parseable JSON, it is usually too small or is a base rather than an
|
|
536
|
+
instruction-tuned model.
|
|
537
|
+
- **Theme induction under-covers dimensions, and which dimension it drops
|
|
538
|
+
varies by model.** Running the shipped 24-transcript corpus through four
|
|
539
|
+
local models gave four different answers: qwen2.5:7b covered 1 of the 4
|
|
540
|
+
dimensions, llama3:8b missed anticipation, mistral:7b missed reflexivity,
|
|
541
|
+
gemma3:12b missed responsiveness — even though the corpus contains 43–49
|
|
542
|
+
codes in every dimension by construction. Do not read a missing dimension
|
|
543
|
+
in the data structure figure as a finding. The Theme structure tab now says
|
|
544
|
+
so explicitly when a dimension receives no themes; check the first-order
|
|
545
|
+
concepts before concluding anything, and prefer the largest model you can
|
|
546
|
+
run.
|
|
547
|
+
- Lexicon quality depends on the segmenter; without CKIP some Chinese entries
|
|
548
|
+
will be wrong word boundaries.
|
|
549
|
+
- Many statistics are meaningless at small *n*. The tool withholds figures whose
|
|
550
|
+
assumptions fail, but it cannot tell you whether twenty-four respondents
|
|
551
|
+
support your conclusion.
|
|
552
|
+
- Framework construction depends on OpenAlex coverage. Niche fields,
|
|
553
|
+
non-English literature and very recent work may not be retrievable.
|
|
554
|
+
- **This tool will not make a study rigorous.** It makes what you did traceable,
|
|
555
|
+
reportable and checkable.
|
|
556
|
+
|
|
557
|
+
---
|
|
558
|
+
|
|
559
|
+
## License
|
|
560
|
+
|
|
561
|
+
MIT — see [LICENSE.txt](LICENSE.txt).
|