tacit-qda 1.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. tacit_qda-1.1.1/LICENSE.txt +21 -0
  2. tacit_qda-1.1.1/PKG-INFO +561 -0
  3. tacit_qda-1.1.1/README.md +531 -0
  4. tacit_qda-1.1.1/frameworks/esg_disclosure_probe.json +226 -0
  5. tacit_qda-1.1.1/frameworks/ri_stilgoe_2013.json +332 -0
  6. tacit_qda-1.1.1/frameworks/tdf_cane_2012.json +594 -0
  7. tacit_qda-1.1.1/frameworks/utaut_venkatesh_2003.json +226 -0
  8. tacit_qda-1.1.1/pyproject.toml +54 -0
  9. tacit_qda-1.1.1/setup.cfg +4 -0
  10. tacit_qda-1.1.1/src/__init__.py +53 -0
  11. tacit_qda-1.1.1/src/_demo_en_academia.py +639 -0
  12. tacit_qda-1.1.1/src/_demo_en_government.py +623 -0
  13. tacit_qda-1.1.1/src/_demo_en_industry.py +636 -0
  14. tacit_qda-1.1.1/src/_demo_en_nonprofit.py +622 -0
  15. tacit_qda-1.1.1/src/_launch_common.py +348 -0
  16. tacit_qda-1.1.1/src/_tacit_cli.py +28 -0
  17. tacit_qda-1.1.1/src/app.py +3781 -0
  18. tacit_qda-1.1.1/src/bench_agreement.py +134 -0
  19. tacit_qda-1.1.1/src/bench_models.py +481 -0
  20. tacit_qda-1.1.1/src/bench_open_coding.py +329 -0
  21. tacit_qda-1.1.1/src/bench_open_summary.py +181 -0
  22. tacit_qda-1.1.1/src/bench_yield.py +140 -0
  23. tacit_qda-1.1.1/src/demo_transcripts_en.py +238 -0
  24. tacit_qda-1.1.1/src/demo_transcripts_zh.py +333 -0
  25. tacit_qda-1.1.1/src/make_demo_data.py +427 -0
  26. tacit_qda-1.1.1/src/make_hearing_corpus.py +547 -0
  27. tacit_qda-1.1.1/src/tacit_analysis.py +428 -0
  28. tacit_qda-1.1.1/src/tacit_coding.py +321 -0
  29. tacit_qda-1.1.1/src/tacit_framework.py +977 -0
  30. tacit_qda-1.1.1/src/tacit_i18n.py +496 -0
  31. tacit_qda-1.1.1/src/tacit_irr.py +586 -0
  32. tacit_qda-1.1.1/src/tacit_lexicon.py +1125 -0
  33. tacit_qda-1.1.1/src/tacit_llm.py +1029 -0
  34. tacit_qda-1.1.1/src/tacit_open.py +755 -0
  35. tacit_qda-1.1.1/src/tacit_openalex.py +1184 -0
  36. tacit_qda-1.1.1/src/tacit_review.py +384 -0
  37. tacit_qda-1.1.1/src/tacit_schema.py +876 -0
  38. tacit_qda-1.1.1/src/tacit_strings.py +1828 -0
  39. tacit_qda-1.1.1/src/tacit_themes.py +1056 -0
  40. tacit_qda-1.1.1/tacit_qda.egg-info/PKG-INFO +561 -0
  41. tacit_qda-1.1.1/tacit_qda.egg-info/SOURCES.txt +63 -0
  42. tacit_qda-1.1.1/tacit_qda.egg-info/dependency_links.txt +1 -0
  43. tacit_qda-1.1.1/tacit_qda.egg-info/entry_points.txt +2 -0
  44. tacit_qda-1.1.1/tacit_qda.egg-info/requires.txt +11 -0
  45. tacit_qda-1.1.1/tacit_qda.egg-info/top_level.txt +1 -0
  46. tacit_qda-1.1.1/tests/test_analysis_engine.py +320 -0
  47. tacit_qda-1.1.1/tests/test_app_functions.py +405 -0
  48. tacit_qda-1.1.1/tests/test_bench_resume.py +146 -0
  49. tacit_qda-1.1.1/tests/test_coding.py +310 -0
  50. tacit_qda-1.1.1/tests/test_demo_data.py +206 -0
  51. tacit_qda-1.1.1/tests/test_framework.py +439 -0
  52. tacit_qda-1.1.1/tests/test_frameworks_shipped.py +293 -0
  53. tacit_qda-1.1.1/tests/test_i18n.py +164 -0
  54. tacit_qda-1.1.1/tests/test_irr.py +406 -0
  55. tacit_qda-1.1.1/tests/test_launcher.py +339 -0
  56. tacit_qda-1.1.1/tests/test_lexicon.py +436 -0
  57. tacit_qda-1.1.1/tests/test_llm_providers.py +961 -0
  58. tacit_qda-1.1.1/tests/test_llm_retry.py +283 -0
  59. tacit_qda-1.1.1/tests/test_open_coding.py +568 -0
  60. tacit_qda-1.1.1/tests/test_openalex.py +635 -0
  61. tacit_qda-1.1.1/tests/test_review.py +253 -0
  62. tacit_qda-1.1.1/tests/test_schema.py +451 -0
  63. tacit_qda-1.1.1/tests/test_themes.py +678 -0
  64. tacit_qda-1.1.1/tests/test_ui_full.py +394 -0
  65. tacit_qda-1.1.1/tests/test_ui_smoke.py +915 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Hung-Chi Chang
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,561 @@
1
+ Metadata-Version: 2.4
2
+ Name: tacit-qda
3
+ Version: 1.1.1
4
+ Summary: Auditable, locally deployable LLM-assisted qualitative coding with codebook provenance
5
+ Author: Hung-Chi Chang
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/NYCU-FUN-TECH-LAB/TACIT
8
+ Project-URL: Documentation, https://github.com/NYCU-FUN-TECH-LAB/TACIT/blob/main/README.md
9
+ Project-URL: Archive, https://doi.org/10.5281/zenodo.22882168
10
+ Keywords: qualitative research,qualitative coding,thematic analysis,large language models,codebook,inter-rater reliability,CAQDAS
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
15
+ Classifier: Topic :: Sociology
16
+ Requires-Python: >=3.9
17
+ Description-Content-Type: text/markdown
18
+ License-File: LICENSE.txt
19
+ Requires-Dist: streamlit>=1.31
20
+ Requires-Dist: python-docx>=1.1
21
+ Requires-Dist: pandas>=2.0
22
+ Requires-Dist: numpy>=1.24
23
+ Requires-Dist: openpyxl>=3.1
24
+ Requires-Dist: google-genai>=1.0
25
+ Requires-Dist: plotly>=5.18
26
+ Requires-Dist: scipy>=1.11
27
+ Provides-Extra: zh
28
+ Requires-Dist: ckip-transformers; extra == "zh"
29
+ Dynamic: license-file
30
+
31
+ # TACIT
32
+
33
+ **Theory-Anchored Coding with Interpretive Transparency.**
34
+
35
+ A Streamlit application that codes interview transcripts against a *pluggable*
36
+ theoretical framework, keeps interpretive authority with the researcher, and
37
+ reports the statistics a reader will actually ask for — including inter-rater
38
+ reliability computed from genuine double-blind double coding.
39
+
40
+ Language models can run **entirely on your own machine**, so transcripts need
41
+ not leave it.
42
+
43
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE.txt)
44
+ [![Python 3.9+](https://img.shields.io/badge/python-3.9%2B-blue.svg)](https://www.python.org/)
45
+ [![Tests](https://img.shields.io/badge/tests-20%20suites-brightgreen.svg)](#testing)
46
+ [![DOI](https://zenodo.org/badge/DOI/10.5281/zenodo.22882168.svg)](https://doi.org/10.5281/zenodo.22882168)
47
+
48
+ ---
49
+
50
+ ## What makes this different
51
+
52
+ Most qualitative tools treat a language model as an autocoder: feed it a
53
+ transcript, get codes back, accept them. This one refuses to work that way, for
54
+ four reasons.
55
+
56
+ **The framework comes from literature, not from the model's memory.**
57
+ A coding framework is the validity foundation of the whole study. Asking a model
58
+ to recall the dimensions of a theory produces output that is plausible and
59
+ partly invented, including citations to work that does not exist. Here you
60
+ retrieve literature from OpenAlex, the model drafts dimensions *from the
61
+ retrieved abstracts*, and a hallucination guard rejects any citation that was
62
+ not actually retrieved. You then approve the framework dimension by dimension,
63
+ under your own name.
64
+
65
+ **Interpretive authority stays with the researcher.**
66
+ Model coding is a draft. Every code can be confirmed, revised or deleted, and
67
+ you can add segments the model missed. Every change is written to an audit
68
+ trail, and the model's original judgement is retained — so "how often did the
69
+ model need correcting?" becomes a reportable number rather than an unknown.
70
+
71
+ **Anything computable is computed, not estimated.**
72
+ Reliability uses a sampling frame that *includes unmarked units*, so recall is
73
+ meaningful — without them, recall is identically 1 and the number is
74
+ meaningless. When the assumptions of a chi-square test fail, the tool refuses to
75
+ print a p-value rather than inviting a claim that cannot survive scrutiny.
76
+
77
+ **Your data does not have to leave your machine.**
78
+ Interview transcripts are human-subject data. Whether they may be sent to a
79
+ third-party service is decided by your ethics approval, and approvals written
80
+ before generative AI generally do not cover it. Point TACIT at a local model
81
+ server and nothing goes over the network. It also helps reproducibility: cloud
82
+ models get withdrawn, and the weights behind a version string change silently,
83
+ whereas a local model has a fixed weight file you can name in a paper.
84
+
85
+ ---
86
+
87
+ ## Try it
88
+
89
+ | | |
90
+ |---|---|
91
+ | **Manual (English)** | `docs/TACIT_User-Manual_en.pdf` |
92
+ | **Manual (中文)** | `docs/TACIT_User-Manual_zh.pdf` |
93
+ | **Demo corpora** | 80 witness transcripts from U.S. congressional hearings on AI governance (public domain) · 24 synthetic English interviews with reference codings · 6 Traditional Chinese |
94
+
95
+ The synthetic interviews ship with **a reference coding for every one of
96
+ them**, so you can open every tab and see real results with **no API key and
97
+ no local model**. The hearing transcripts are real speech you can check against
98
+ its source. See [Demonstration data](#demonstration-data).
99
+
100
+ ---
101
+
102
+ ## Evaluating TACIT without a model
103
+
104
+ You do not need a model to evaluate this software. There are three levels, and
105
+ the first two need nothing installed beyond the requirements.
106
+
107
+ **Level 1 — no model, no key, our data.** Install, launch, and in the sidebar
108
+ tick the shipped analyses under `analyses/`. Every analysis tab then works on
109
+ real coded material: cross-tabulations, co-occurrence, the cross-case matrix,
110
+ the audit trail in the coding-review tab, the reliability report, and the
111
+ Excel/Gioia exports. This exercises everything except the calls to the model
112
+ itself. The shipped codings are read-only — editing one saves a new file and
113
+ records what it was derived from, so the reference material cannot be
114
+ overwritten by accident.
115
+
116
+ **Level 2 — no model, no key, your own transcripts.** Open the reliability tab
117
+ and set *Build the frame from* to *Transcripts I upload here*. Upload your own
118
+ `.docx` files; TACIT splits them into speaking units, draws a blinded sample
119
+ from a seed you choose, and issues one coding sheet per coder. Take the
120
+ completed sheets back and you get percentage agreement, Cohen's κ, PABAK,
121
+ Gwet's AC1 and Krippendorff's α, per code and pooled, with the disagreement
122
+ list.
123
+
124
+ That is not the end of it. *Turn these codings into an analysis* sends a
125
+ coder's completed sheet into the workspace as ordinary records, and every
126
+ analysis tab then works on them: crosstabs by case descriptor with the
127
+ chi-square guard, co-occurrence, the cross-case matrix, lexicon discovery, the
128
+ Excel and Gioia exports — and theme grouping, which clusters the codes by
129
+ co-occurrence or by label similarity and leaves the naming to you. A complete
130
+ two-coder thematic analysis, from transcript to data structure, with no model
131
+ involved at any point and nothing leaving the machine.
132
+
133
+ **Level 3 — with a model.** Either a free Google Gemini key
134
+ (<https://aistudio.google.com/apikey>, no payment details required) or a local
135
+ Ollama model. See [Choosing a model provider](#choosing-a-model-provider).
136
+
137
+ One thing about the free Gemini tier is worth knowing before you plan a run,
138
+ because it is easy to misread as a broken key. **The daily request quota is per
139
+ project and per model, not per key**, so issuing a second key in the same
140
+ project changes nothing. The limit differs sharply between models: as of
141
+ September 2026 `gemini-3.6-flash` allowed 20 requests per day, which will not
142
+ finish a single transcript, while `gemini-3.5-flash-lite` completed a
143
+ 12-transcript windowed run (149 requests) with room to spare. A 429 reply names
144
+ the quota it hit — look for `GenerateRequestsPerDayPerProjectPerModel` and the
145
+ `limit:` value — and TACIT reports quota errors separately from a busy server,
146
+ because waiting helps with the second and not with the first.
147
+
148
+ ---
149
+
150
+ ## Install
151
+
152
+ Requires Python 3.9+.
153
+
154
+ ```bash
155
+ git clone https://github.com/NYCU-FUN-TECH-LAB/TACIT.git
156
+ cd TACIT
157
+ ```
158
+
159
+ Then double-click the launcher for your platform:
160
+
161
+ | Platform | Launcher |
162
+ |---|---|
163
+ | Windows | `launch_windows.bat` |
164
+ | macOS | `launch_mac.command` |
165
+ | Linux | `launch_linux.sh` |
166
+
167
+ The launcher creates an isolated environment, installs dependencies, picks a
168
+ free port and opens a browser. First run takes 3–10 minutes.
169
+
170
+ Prefer to do it manually:
171
+
172
+ ```bash
173
+ python -m venv .venv && source .venv/bin/activate # Windows: .venv\Scripts\activate
174
+ pip install -r requirements.txt
175
+ streamlit run src/app.py
176
+ ```
177
+
178
+ ### Install from PyPI
179
+
180
+ TACIT is also published as a Python package:
181
+
182
+ ```bash
183
+ pip install tacit-qda
184
+ tacit-qda
185
+ ```
186
+
187
+ `tacit-qda` starts the interface with the current directory as the workspace:
188
+ frameworks, analyses and lexicons are read from and written there, and the four
189
+ shipped frameworks are copied into `frameworks/` on first use. Arguments after
190
+ the command go to `streamlit run` (for example `tacit-qda --server.port 8600`).
191
+ The demonstration corpora and benchmark scripts are in this repository, not in
192
+ the package. `pip install "tacit-qda[zh]"` adds the Traditional Chinese segmenter.
193
+
194
+ The engines can also be used from a script:
195
+
196
+ ```python
197
+ import tacit_qda # puts the TACIT modules on the import path
198
+ import tacit_framework as F
199
+ F.FRAMEWORK_DIR = tacit_qda.bundled_frameworks_dir()
200
+ F.activate_by_id("ri_stilgoe_2013")
201
+ ```
202
+
203
+ ---
204
+
205
+ ## Choosing a model provider
206
+
207
+ Pick the provider at the top of the sidebar. Everything downstream — coding,
208
+ theme induction, framework drafting — uses whichever you chose.
209
+
210
+ | Provider | Needs a key | Data leaves your machine | Extra Python package |
211
+ |---|---|---|---|
212
+ | **Ollama** | no | **no** | none |
213
+ | **OpenAI-compatible endpoint** (LM Studio, llama.cpp server, vLLM, LocalAI) | no, if it is on localhost | **no**, if it is on localhost | none |
214
+ | **Google Gemini** | yes | yes | `google-genai` (installed with the requirements) |
215
+
216
+ All three providers work straight after installation. The Google SDK is part
217
+ of `requirements.txt`, so a machine without Ollama or any other local model
218
+ server can use Gemini as soon as TACIT is installed. The local providers use
219
+ only the Python standard library.
220
+
221
+ The older `google-generativeai` package still works and is detected
222
+ automatically, but Google has ended support for it; the sidebar will say so and
223
+ point at the replacement.
224
+
225
+ ### Local setup in three commands
226
+
227
+ ```bash
228
+ ollama serve
229
+ ollama pull llama3.1:8b-instruct-q4_K_M
230
+ streamlit run src/app.py
231
+ ```
232
+
233
+ Then choose **Ollama (local)** in the sidebar, set the address to
234
+ `http://localhost:11434`, and press **Test connection**.
235
+
236
+ ### One thing you must get right: the context window
237
+
238
+ Ollama's default context window is 4,096 tokens. **One interview does not fit.**
239
+ A server that runs out of context truncates the input *silently* and still
240
+ returns well-formed JSON — so your coding looks complete while half the
241
+ transcript was never read. This is the most dangerous failure mode in local
242
+ deployment precisely because nothing goes wrong on screen.
243
+
244
+ TACIT therefore sets the context window explicitly (default 32,768), exposes it
245
+ in the sidebar, estimates the prompt length before sending, and **refuses to
246
+ send a prompt that would not fit**. Raise it for long transcripts; it costs
247
+ memory, roughly 0.5–2 GB per 32k tokens depending on the model.
248
+
249
+ ### Batch and scripted runs
250
+
251
+ ```bash
252
+ export TACIT_PROVIDER=ollama
253
+ export TACIT_MODEL=llama3.1:8b-instruct-q4_K_M
254
+ export TACIT_BASE_URL=http://localhost:11434
255
+ export TACIT_NUM_CTX=32768
256
+ ```
257
+
258
+ The sidebar picks these up as defaults.
259
+
260
+ ### What gets recorded
261
+
262
+ Every analysis record stores the **full endpoint descriptor**, not just the
263
+ model name — `ollama/llama3.1:8b-instruct-q4_K_M@http://localhost:11434` rather
264
+ than `llama3.1:8b`. The same short name means different quantisations on
265
+ different machines, and a record that only names the model cannot be
266
+ reproduced.
267
+
268
+ ---
269
+
270
+ ## Pluggable frameworks
271
+
272
+ Three frameworks ship with the tool, and they are deliberately unalike — each
273
+ one exercises a different part of the framework contract:
274
+
275
+ | Framework | Dimensions | Polarity | Field | What it tests |
276
+ |---|---|---|---|---|
277
+ | `ri_stilgoe_2013` | Anticipation, Reflexivity, Engagement, Responsiveness | yes, `P`/`N` → 8 codes | Responsible innovation | the default |
278
+ | `utaut_venkatesh_2003` | Performance expectancy, Effort expectancy, Social influence, Facilitating conditions | none → 4 codes | Technology acceptance | a framework with **no polarity model** |
279
+ | `esg_disclosure_probe` | Targets and baselines, Measurement and boundary, Governance and accountability, Stakeholder engagement | yes, `S`/`A` → 8 codes | Sustainability disclosure | polarity values that are **not** `P`/`N`; a non-interview corpus |
280
+
281
+ None of them is hard-coded. Dimensions, polarity values, descriptor fields and
282
+ interface labels live in a JSON file, and the analysis engines, prompts and
283
+ schema all derive from whichever framework is active. Switching to UTAUT
284
+ collapses codes from `ANT-P`/`ANT-N` to `PE`, and the polarity sub-tab
285
+ disappears because that analysis is meaningless without a polarity model.
286
+
287
+ The third framework guards against a specific failure: code that derives the *dimensions* from the active framework but still wrote the two
288
+ polarity values as the literals `"P"` and `"N"` in the polarity-balance
289
+ statistics and in the lexicon coder — which no test caught, because the only
290
+ other framework that shipped had no polarity model at all and skipped those code
291
+ paths entirely. `esg_disclosure_probe` names its poles `S` (substantiated) and
292
+ `A` (asserted), so it does exercise them. `tests/test_frameworks_shipped.py`
293
+ runs every framework in `frameworks/` through the full analysis pipeline for
294
+ exactly this reason: a pluggability claim is only worth as much as the least
295
+ similar framework you have actually run.
296
+
297
+ Any deductive coding scheme is in scope — template analysis, framework analysis,
298
+ theory-driven content analysis. Write a JSON file, or build one from retrieved
299
+ literature in the Framework builder tab.
300
+
301
+ ---
302
+
303
+ ## Demonstration data
304
+
305
+ Two corpora ship, and they are different kinds of thing. Do not mix them up.
306
+
307
+ | | Congressional hearings | Synthetic interviews | Synthetic, Chinese |
308
+ |---|---|---|---|
309
+ | Folder | `demo_data/hearings/` | `demo_data/en/` | `demo_data/zh/` |
310
+ | What it is | **real** testimony, public record | **fictional**, written by a language model | fictional |
311
+ | Transcripts | 80 witnesses, 22 hearings (24 form the paper's subset) | 24 | 6 |
312
+ | Sectors | industry, academia, government, civil society | the same four, six each | 4 |
313
+ | Reference codings | none | 24, in `analyses/` | — |
314
+ | Use it for | open coding, screenshots, anything you want a reader to be able to verify | statistics, agreement against a fixed standard, regression tests | CKIP segmentation, language routing |
315
+
316
+ **The hearings** are U.S. congressional hearings on artificial-intelligence
317
+ governance, March 2023 to June 2025, retrieved from
318
+ [govinfo.gov](https://www.govinfo.gov) and in the public domain under
319
+ 17 U.S.C. § 105. The record was split at witness level, one file per speaker;
320
+ prepared statements reprinted in the record were left out because they are
321
+ written documents, not testimony. `manifest.json` gives the package identifier,
322
+ hearing title, date, chamber and source URL for every witness, the hand-assigned
323
+ sector, and the three witnesses excluded with the reason for each.
324
+
325
+ ```bash
326
+ python src/make_hearing_corpus.py # re-fetches from govinfo and rebuilds the corpus
327
+ ```
328
+
329
+ **The synthetic interviews** are fictional. Transcripts and reference codings
330
+ alike were written by a language model (Claude) against the responsible-innovation
331
+ codebook, so every quotation is by construction a clean substring of a passage
332
+ that exemplifies its code. That is what makes them a fixed comparison standard,
333
+ and it is also why agreement measured against them is an **upper bound** —
334
+ expect lower figures on real transcripts. They must not be cited as empirical
335
+ data. Read `demo_data/en/00_ABOUT_THIS_DATA.txt` before using them.
336
+
337
+ ```bash
338
+ python src/make_demo_data.py # rebuilds the .docx files and reference codings
339
+ ```
340
+
341
+ ### Why twenty-four and not six
342
+
343
+ Six transcripts are enough to walk through the interface and not enough for a
344
+ single inferential statistic to run. In a 2×4 crosstab of institution type
345
+ against polarity, expected cell counts land around one or two, the chi-square
346
+ assumptions fail, and the tool correctly refuses to print a p-value. A newcomer
347
+ then opens the cross-analysis tab and sees a column of insufficient-data
348
+ notices — the part of the tool most worth examining shows nothing at all.
349
+
350
+ With 24 respondents the corpus produces:
351
+
352
+ | Quantity | Value |
353
+ |---|---|
354
+ | Coded segments | 148, of which 36 (24%) are coded more than once |
355
+ | Codes assigned | 184 (one per researcher judgement) |
356
+ | Analysis units | 181 distinct segment × dimension × polarity |
357
+ | Reliability sampling frame | 299 units, **174 of them unmarked** |
358
+ | Institution type × polarity | χ²(3) = 10.04, *p* = .018, *V* = .236, min. expected 18.1 |
359
+ | Institution type × dimension | χ²(9) = 7.19, *p* = .617, *V* = .115, assumptions met |
360
+ | Institution type × full code set | assumptions not met, p value withheld |
361
+
362
+ Those three rows are all deliberate: one significant, one null with assumptions
363
+ met, one withheld. A demonstration corpus in which every test came out
364
+ significant would be demonstrating the corpus, not the method.
365
+
366
+ The two code counts differ because they answer different questions. Three
367
+ segments carry two judgements on the same dimension and polarity, with
368
+ different rationales; the record keeps both, because a rationale is the
369
+ researcher's reasoning and not an annotation. The analysis layer counts each
370
+ segment once per dimension and polarity, because counting one passage twice
371
+ under the same code would inflate its co-occurrence and Jaccard weights.
372
+ Both numbers are correct; any report has to say which one it is using.
373
+
374
+ ### About the reference codings
375
+
376
+ The codings in `analyses/` were **written alongside the transcripts**. They are
377
+ not the output of any particular model run and should not be read as evidence of
378
+ how well any model performs. They exist so every tab can be opened without an
379
+ API key.
380
+
381
+ Using them as a comparison set for reliability *is* a legitimate use: run the
382
+ coder yourself, then compare your run against this fixed, deliberately designed
383
+ scheme.
384
+
385
+ ---
386
+
387
+ ## Handling real interview data
388
+
389
+ **Keep confidential transcripts outside this folder.** `.gitignore` only
390
+ excludes files that are not already tracked; it is a safety net, not a
391
+ guarantee. A single `git add -f` defeats it.
392
+
393
+ ```
394
+ project/
395
+ ├── tacit/ ← this repository
396
+ └── research-data/ ← your transcripts and analyses, never committed
397
+ ```
398
+
399
+ If your ethics approval does not permit sending transcripts to a third-party
400
+ service, use a local provider and verify the address is on `localhost`. The
401
+ sidebar states which case you are in.
402
+
403
+ ---
404
+
405
+ ## What's in each tab
406
+
407
+ | Tab | Purpose |
408
+ |---|---|
409
+ | Run analysis | Code transcripts against the active framework. The only step that uses model capacity |
410
+ | Data & descriptors | Edit respondent attributes; these drive every later comparison |
411
+ | Code review | Confirm, revise, delete or add codes. Full audit trail |
412
+ | Cross-analysis | Crosstabs, co-occurrence, cross-case matrix, polarity balance |
413
+ | Theme structure | Two-stage inductive theme induction with a model, or model-free grouping of the codes by co-occurrence or label similarity which you name yourself; Gioia data structure figure with SVG export |
414
+ | Lexicon induction | Term discovery, log-odds feature induction, dictionary baseline for auditing the model |
415
+ | Reliability | Sampling frame from records or from transcripts you upload, double-blind coding sheets, κ / PABAK / AC1 / Krippendorff α, confusion matrices, model precision and recall, and human codings turned back into analysable records |
416
+ | Export | Excel, Word, JSON |
417
+ | Framework builder | Retrieve literature from OpenAlex, draft dimensions, approve them individually |
418
+
419
+ ---
420
+
421
+ ## Reproducing the numbers in the paper
422
+
423
+ Every figure quoted in the SoftwareX article comes from one of the scripts
424
+ below, and each writes its raw output under `bench_out/`, which is kept in this
425
+ repository. Run them from the project root with the virtual environment active.
426
+
427
+ | What it produces | Command |
428
+ |---|---|
429
+ | Demonstration corpus and reference codings (Table 7) | `python src/make_demo_data.py` then `python run_tests.py demo` |
430
+ | Model comparison: segments, codes, κ against the reference codings, wall time (Table 6) | `python src/bench_models.py --provider ollama --models llama3:8b` |
431
+ | The same for a cloud model | `python src/bench_models.py --provider gemini --models gemini-3.5-flash-lite --api-key <key>` |
432
+ | Agreement against the reference codings on the 299-unit frame (Section 3) | `python src/bench_agreement.py bench_out/local_8b/records/llama3_8b/en` |
433
+ | Single-pass against windowed yield (Section 2.4) | `python src/bench_yield.py --model llama3:8b` |
434
+ | Open coding, one run on the twelve-transcript subset (Table 8) | `python src/bench_open_coding.py --corpus hearings --model llama3:8b --per-sector 3 --tag 12_run1` |
435
+ | The same with a cloud model | `python src/bench_open_coding.py --corpus hearings --provider gemini --model gemini-3.5-flash-lite --per-sector 3 --tag 12_run1` |
436
+ | Table 8 itself: median and range over the archived runs, no model needed | `python src/bench_open_summary.py` |
437
+ | Figure 2, computed from the shipped context formula | `python paper/make_fig2.py` |
438
+ | Figure 6, the three-level data structure | `python paper/make_fig6.py` |
439
+
440
+ Two caveats about exactness.
441
+
442
+ **The cloud row cannot be reproduced identically.** Weights behind a stable
443
+ version string change and older versions are withdrawn, which is one of the
444
+ arguments the article makes. The archived outputs under
445
+ `bench_out/cloud/` are therefore the record of what that run did; re-running
446
+ gives a comparable but not identical result.
447
+
448
+ **Runs vary between samples, by a lot.** On the same twelve transcripts, in the
449
+ same order, at the same temperature, the cloud model returned 89, 143 and 156
450
+ codes and the local 8B model 11, 27 and 108. One open-coding run is therefore
451
+ not a reportable number. Give each run its own `--tag` (the archived ones are
452
+ `12_run1`, `12_run2`, `12_run3`) so it lands in its own directory; `bench_open_summary.py` then reports the median and range, and
453
+ leaves out runs made by a different version of the code, saying which and why.
454
+
455
+ ---
456
+
457
+ ## Testing
458
+
459
+ ```bash
460
+ python run_tests.py
461
+ ```
462
+
463
+ Seventeen suites, **none requiring network access, an API key or a model
464
+ server**. The OpenAlex client is tested against offline fixtures, the provider
465
+ abstraction against fake transports, and the interface end-to-end with
466
+ Streamlit's `AppTest` in both languages and several frameworks.
467
+
468
+ `tests/test_frameworks_shipped.py` enumerates the real `frameworks/` directory
469
+ rather than a hard-coded list, and pushes every framework it finds through the
470
+ whole pipeline — schema, prompt, cross-tabs, co-occurrence, polarity balance,
471
+ lexicon, review, themes, reliability sampling and interface labels. Adding a
472
+ framework file adds it to the test run automatically. This suite exists because
473
+ the pluggability claim had been tested only against frameworks that happened to
474
+ resemble the default one.
475
+
476
+ Run one suite:
477
+
478
+ ```bash
479
+ python run_tests.py demo
480
+ ```
481
+
482
+ ---
483
+
484
+ ## Optional: Traditional Chinese segmentation
485
+
486
+ ```bash
487
+ pip install ckip-transformers
488
+ ```
489
+
490
+ Installed automatically by the launchers. Without it the lexicon module falls
491
+ back to an n-gram heuristic with no linguistic knowledge of Chinese, and some
492
+ entries will be wrong word boundaries. Semantic coding does not pass through
493
+ segmentation and is unaffected.
494
+
495
+ ---
496
+
497
+ ## Citation
498
+
499
+ If this tool contributes to published work, please cite it. See
500
+ [CITATION.cff](CITATION.cff); machine-readable metadata is in
501
+ [codemeta.json](codemeta.json).
502
+
503
+ Every release is archived on Zenodo. Version 1.0.0 is
504
+ [10.5281/zenodo.22882169](https://doi.org/10.5281/zenodo.22882169);
505
+ [10.5281/zenodo.22882168](https://doi.org/10.5281/zenodo.22882168) always resolves to
506
+ the latest version.
507
+
508
+ The built-in frameworks rest on:
509
+
510
+ - Stilgoe, J., Owen, R., & Macnaghten, P. (2013). Developing a framework for
511
+ responsible innovation. *Research Policy, 42*(9), 1568–1580.
512
+ https://doi.org/10.1016/j.respol.2013.05.008
513
+ - Venkatesh, V., Morris, M. G., Davis, G. B., & Davis, F. D. (2003). User
514
+ acceptance of information technology: Toward a unified view. *MIS Quarterly,
515
+ 27*(3), 425–478. https://doi.org/10.2307/30036540
516
+
517
+ Statistical methods:
518
+
519
+ - Monroe, B. L., Colaresi, M. P., & Quinn, K. M. (2008). Fightin' words.
520
+ *Political Analysis, 16*(4), 372–403.
521
+ - Gwet, K. L. (2008). Computing inter-rater reliability and its variance in the
522
+ presence of high agreement. *British Journal of Mathematical and Statistical
523
+ Psychology, 61*(1), 29–48.
524
+ - Gioia, D. A., Corley, K. G., & Hamilton, A. L. (2013). Seeking qualitative
525
+ rigor in inductive research. *Organizational Research Methods, 16*(1), 15–31.
526
+
527
+ ---
528
+
529
+ ## Limitations
530
+
531
+ - Model coding is a draft, not ground truth. Unreviewed coding should not be
532
+ written up as verified.
533
+ - Small local models produce noticeably fewer segments and follow the output
534
+ format less reliably than large ones. If a model keeps failing to return
535
+ parseable JSON, it is usually too small or is a base rather than an
536
+ instruction-tuned model.
537
+ - **Theme induction under-covers dimensions, and which dimension it drops
538
+ varies by model.** Running the shipped 24-transcript corpus through four
539
+ local models gave four different answers: qwen2.5:7b covered 1 of the 4
540
+ dimensions, llama3:8b missed anticipation, mistral:7b missed reflexivity,
541
+ gemma3:12b missed responsiveness — even though the corpus contains 43–49
542
+ codes in every dimension by construction. Do not read a missing dimension
543
+ in the data structure figure as a finding. The Theme structure tab now says
544
+ so explicitly when a dimension receives no themes; check the first-order
545
+ concepts before concluding anything, and prefer the largest model you can
546
+ run.
547
+ - Lexicon quality depends on the segmenter; without CKIP some Chinese entries
548
+ will be wrong word boundaries.
549
+ - Many statistics are meaningless at small *n*. The tool withholds figures whose
550
+ assumptions fail, but it cannot tell you whether twenty-four respondents
551
+ support your conclusion.
552
+ - Framework construction depends on OpenAlex coverage. Niche fields,
553
+ non-English literature and very recent work may not be retrievable.
554
+ - **This tool will not make a study rigorous.** It makes what you did traceable,
555
+ reportable and checkable.
556
+
557
+ ---
558
+
559
+ ## License
560
+
561
+ MIT — see [LICENSE.txt](LICENSE.txt).