batchalign 0.8.2.post11__tar.gz → 0.8.2.post12__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. {batchalign-0.8.2.post11/batchalign.egg-info → batchalign-0.8.2.post12}/PKG-INFO +1 -1
  2. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/cli/dispatch.py +1 -0
  3. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/chat/file.py +4 -3
  4. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/chat/generator.py +3 -3
  5. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/analysis/compare.py +21 -21
  6. batchalign-0.8.2.post12/batchalign/version +3 -0
  7. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12/batchalign.egg-info}/PKG-INFO +1 -1
  8. batchalign-0.8.2.post11/batchalign/version +0 -3
  9. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/LICENSE +0 -0
  10. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/MANIFEST.in +0 -0
  11. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/README.md +0 -0
  12. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/__init__.py +0 -0
  13. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/__main__.py +0 -0
  14. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/cli/__init__.py +0 -0
  15. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/cli/bench.py +0 -0
  16. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/cli/cache.py +0 -0
  17. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/cli/cli.py +0 -0
  18. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/constants.py +0 -0
  19. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/document.py +0 -0
  20. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/errors.py +0 -0
  21. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/__init__.py +0 -0
  22. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/base.py +0 -0
  23. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/chat/__init__.py +0 -0
  24. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/chat/lexer.py +0 -0
  25. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/chat/parser.py +0 -0
  26. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/chat/utils.py +0 -0
  27. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/textgrid/__init__.py +0 -0
  28. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/textgrid/file.py +0 -0
  29. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/textgrid/generator.py +0 -0
  30. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/formats/textgrid/parser.py +0 -0
  31. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/__init__.py +0 -0
  32. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/audio_io.py +0 -0
  33. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/resolve.py +0 -0
  34. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/speaker/__init__.py +0 -0
  35. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/speaker/config.yaml +0 -0
  36. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/speaker/infer.py +0 -0
  37. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/speaker/utils.py +0 -0
  38. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/training/__init__.py +0 -0
  39. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/training/run.py +0 -0
  40. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/training/utils.py +0 -0
  41. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/utils.py +0 -0
  42. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/utterance/__init__.py +0 -0
  43. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/utterance/cantonese_infer.py +0 -0
  44. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/utterance/dataset.py +0 -0
  45. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/utterance/execute.py +0 -0
  46. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/utterance/infer.py +0 -0
  47. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/utterance/prep.py +0 -0
  48. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/utterance/train.py +0 -0
  49. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/wave2vec/__init__.py +0 -0
  50. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/wave2vec/infer_fa.py +0 -0
  51. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/whisper/__init__.py +0 -0
  52. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/whisper/infer_asr.py +0 -0
  53. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/models/whisper/infer_fa.py +0 -0
  54. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/__init__.py +0 -0
  55. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/analysis/__init__.py +0 -0
  56. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/analysis/eval.py +0 -0
  57. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/__init__.py +0 -0
  58. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2chinese.py +0 -0
  59. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/__init__.py +0 -0
  60. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/deu.py +0 -0
  61. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/ell.py +0 -0
  62. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/eng.py +0 -0
  63. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/eus.py +0 -0
  64. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/fra.py +0 -0
  65. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/hrv.py +0 -0
  66. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/ind.py +0 -0
  67. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/jpn.py +0 -0
  68. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/nld.py +0 -0
  69. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/por.py +0 -0
  70. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/spa.py +0 -0
  71. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/num2lang/tha.py +0 -0
  72. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/oai_whisper.py +0 -0
  73. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/rev.py +0 -0
  74. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/utils.py +0 -0
  75. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/whisper.py +0 -0
  76. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/asr/whisperx.py +0 -0
  77. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/avqi/__init__.py +0 -0
  78. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/avqi/engine.py +0 -0
  79. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/base.py +0 -0
  80. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/cache.py +0 -0
  81. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/cleanup/__init__.py +0 -0
  82. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/cleanup/cleanup.py +0 -0
  83. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/cleanup/disfluencies.py +0 -0
  84. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/cleanup/parse_support.py +0 -0
  85. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/cleanup/retrace.py +0 -0
  86. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/cleanup/support/filled_pauses.eng +0 -0
  87. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/cleanup/support/replacements.eng +0 -0
  88. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/cleanup/support/test.test +0 -0
  89. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/diarization/__init__.py +0 -0
  90. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/diarization/pyannote.py +0 -0
  91. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/dispatch.py +0 -0
  92. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/fa/__init__.py +0 -0
  93. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/fa/wave2vec_fa.py +0 -0
  94. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/fa/whisper_fa.py +0 -0
  95. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/morphosyntax/__init__.py +0 -0
  96. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/morphosyntax/coref.py +0 -0
  97. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/morphosyntax/en/irr.py +0 -0
  98. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/morphosyntax/fr/apm.py +0 -0
  99. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/morphosyntax/fr/apmn.py +0 -0
  100. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/morphosyntax/fr/case.py +0 -0
  101. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/morphosyntax/ja/verbforms.py +0 -0
  102. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/morphosyntax/ud.py +0 -0
  103. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/opensmile/__init__.py +0 -0
  104. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/opensmile/engine.py +0 -0
  105. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/pipeline.py +0 -0
  106. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/segmentation/__init__.py +0 -0
  107. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/segmentation/cantonese_seg.py +0 -0
  108. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/speaker/__init__.py +0 -0
  109. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/speaker/nemo_speaker.py +0 -0
  110. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/translate/__init__.py +0 -0
  111. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/translate/gtrans.py +0 -0
  112. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/translate/seamless.py +0 -0
  113. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/translate/utils.py +0 -0
  114. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/utr/__init__.py +0 -0
  115. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/utr/rev_utr.py +0 -0
  116. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/utr/utils.py +0 -0
  117. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/utr/whisper_utr.py +0 -0
  118. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/utterance/__init__.py +0 -0
  119. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/pipelines/utterance/ud_utterance.py +0 -0
  120. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/__init__.py +0 -0
  121. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/cli/test_dispatch_memory.py +0 -0
  122. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/conftest.py +0 -0
  123. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/formats/chat/test_chat_file.py +0 -0
  124. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/formats/chat/test_chat_generator.py +0 -0
  125. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/formats/chat/test_chat_lexer.py +0 -0
  126. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/formats/chat/test_chat_parser.py +0 -0
  127. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/formats/chat/test_chat_utils.py +0 -0
  128. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/formats/textgrid/test_textgrid.py +0 -0
  129. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/models/test_audio_io.py +0 -0
  130. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/models/test_audio_lazy.py +0 -0
  131. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/analysis/test_eval.py +0 -0
  132. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/asr/test_asr_pipeline.py +0 -0
  133. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/asr/test_asr_utils.py +0 -0
  134. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/cache/__init__.py +0 -0
  135. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/cache/test_cache.py +0 -0
  136. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/cleanup/test_disfluency.py +0 -0
  137. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/cleanup/test_parse_support.py +0 -0
  138. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/fa/test_fa_pipeline.py +0 -0
  139. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/fa/test_fa_short_segments.py +0 -0
  140. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/fixures.py +0 -0
  141. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/test_pipeline.py +0 -0
  142. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/pipelines/test_pipeline_models.py +0 -0
  143. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/tests/test_document.py +0 -0
  144. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/utils/__init__.py +0 -0
  145. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/utils/abbrev.py +0 -0
  146. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/utils/compounds.py +0 -0
  147. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/utils/config.py +0 -0
  148. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/utils/device.py +0 -0
  149. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/utils/dp.py +0 -0
  150. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/utils/names.py +0 -0
  151. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign/utils/utils.py +0 -0
  152. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign.egg-info/SOURCES.txt +0 -0
  153. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign.egg-info/dependency_links.txt +0 -0
  154. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign.egg-info/entry_points.txt +0 -0
  155. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign.egg-info/requires.txt +0 -0
  156. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/batchalign.egg-info/top_level.txt +0 -0
  157. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/setup.cfg +0 -0
  158. {batchalign-0.8.2.post11 → batchalign-0.8.2.post12}/setup.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: batchalign
3
- Version: 0.8.2.post11
3
+ Version: 0.8.2.post12
4
4
  Summary: Python Speech Language Sample Analysis
5
5
  Author: Brian MacWhinney, Houjun Liu
6
6
  Author-email: macw@cmu.edu, houjun@cmu.edu
@@ -226,6 +226,7 @@ def _run_pipeline_for_file(command, pipeline, file, output, loader_info, writer_
226
226
 
227
227
  # Write annotated CHAT
228
228
  CHATFile(doc=result["doc"]).write(output,
229
+ write_mor=False,
229
230
  merge_abbrev=local_kwargs.get("merge_abbrev", False))
230
231
 
231
232
  # Write metrics as JSON sidecar for consolidated CSV later
@@ -93,7 +93,7 @@ class CHATFile(BaseFormat):
93
93
  self.__doc = doc
94
94
 
95
95
 
96
- def write(self, path, write_wor=True, merge_abbrev=False):
96
+ def write(self, path, write_wor=True, write_mor=True, merge_abbrev=False):
97
97
  """Write the CHATFile to file.
98
98
 
99
99
  Parameters
@@ -102,13 +102,13 @@ class CHATFile(BaseFormat):
102
102
  Path of where the CHAT file should get str.
103
103
  """
104
104
 
105
- str_doc = self.__generate(self.__doc, self.__special_mor, write_wor=write_wor, merge_abbrev=merge_abbrev)
105
+ str_doc = self.__generate(self.__doc, self.__special_mor, write_wor=write_wor, write_mor=write_mor, merge_abbrev=merge_abbrev)
106
106
 
107
107
  with open(path, 'w', encoding="utf-8") as df:
108
108
  df.write(str_doc)
109
109
 
110
110
  @staticmethod
111
- def __generate(doc:Document, special=False, write_wor=True, merge_abbrev=False):
111
+ def __generate(doc:Document, special=False, write_wor=True, write_mor=True, merge_abbrev=False):
112
112
  utterances = doc.content
113
113
 
114
114
  def __get_birthdays(line):
@@ -129,6 +129,7 @@ class CHATFile(BaseFormat):
129
129
  else:
130
130
  main.append(generate_chat_utterance(i, special and doc.langs[0] == "eng",
131
131
  write_wor=write_wor,
132
+ write_mor=write_mor,
132
133
  merge_abbrev=merge_abbrev))
133
134
  main.append("@End\n")
134
135
 
@@ -11,7 +11,7 @@ import warnings
11
11
  # document[3].text = None
12
12
  # document[3].model_dump()
13
13
 
14
- def generate_chat_utterance(utterance: Utterance, special_mor=False, write_wor=True, merge_abbrev=False):
14
+ def generate_chat_utterance(utterance: Utterance, special_mor=False, write_wor=True, write_mor=True, merge_abbrev=False):
15
15
  """Converts at Utterance to a CHAT string.
16
16
 
17
17
  Parameters
@@ -89,7 +89,7 @@ def generate_chat_utterance(utterance: Utterance, special_mor=False, write_wor=T
89
89
  # I'm sorry. This is just mor line generation; there is actually not that much complexity here
90
90
  mor_elems.append("~".join(f"{m.pos}|{m.lemma}{'-' if any([m.feats.startswith(i) for i in UD__GENDERS]) else ('-' if m.feats else '')}{m.feats}"
91
91
  for m in mor))
92
- if len(mor_elems) > 0:
92
+ if write_mor and len(mor_elems) > 0:
93
93
  # if the end is punct, drop the tag
94
94
  if mor_elems[-1].startswith("PUNCT"):
95
95
  mor_elems[-1] = mor_elems[-1].split("|")[1]
@@ -97,7 +97,7 @@ def generate_chat_utterance(utterance: Utterance, special_mor=False, write_wor=T
97
97
 
98
98
  #### GRA LINE GENERATION ####
99
99
  # gra list is not different for MWT tokens so we flatten it
100
- if len(gras) > 0:
100
+ if write_mor and len(gras) > 0:
101
101
  flat_gras = [i for j in gras if j for i in j]
102
102
  if flat_gras:
103
103
  result.append(f"%{'u' if special_mor else ''}gra:\t"+" ".join([f"{i.id}|{i.dep_id}|{i.dep_type}" for i in flat_gras]))
@@ -159,10 +159,11 @@ def _find_best_segment(gold_tokens, main_tokens, mfn):
159
159
  multiset overlap with the gold utterance, ignoring order. To keep common
160
160
  words from swallowing later transcript material, it only considers windows
161
161
  near the gold utterance length. Among equally good windows it prefers the
162
- one with the most positional (in-order) matches, then the latest one as a
163
- final tiebreaker (so CHI repetitions / self-corrections beat earlier INV
164
- tokens). The caller then runs the full Levenshtein aligner inside that
165
- window to produce token annotations.
162
+ one with the fewest non-matching (wasted) tokens, then the most
163
+ Levenshtein alignment matches, then the latest one as a final tiebreaker
164
+ (so CHI repetitions / self-corrections beat earlier INV tokens). The caller then
165
+ runs the full Levenshtein aligner inside that window to produce token
166
+ annotations.
166
167
  """
167
168
  if not gold_tokens or not main_tokens:
168
169
  return 0, 0
@@ -176,8 +177,8 @@ def _find_best_segment(gold_tokens, main_tokens, mfn):
176
177
 
177
178
  best = (0, min(main_len, gold_len))
178
179
  best_score = -1.0
179
- best_len_delta = None
180
- best_pos_matches = -1
180
+ best_waste = None
181
+ best_align_matches = -1
181
182
 
182
183
  for span in range(min_window, max_window + 1):
183
184
  window_counts = Counter(main_tokens[:span])
@@ -197,30 +198,29 @@ def _find_best_segment(gold_tokens, main_tokens, mfn):
197
198
  overlap += min(window_counts[right], gold_counts[right])
198
199
 
199
200
  score = overlap / gold_len
200
- len_delta = abs(span - gold_len)
201
+ waste = span - overlap # non-matching tokens in the window
201
202
  end = start + span
202
203
 
203
- # Count how many tokens match the gold in the same position
204
- pos_matches = sum(
205
- 1 for k in range(min(span, gold_len))
206
- if mfn(main_tokens[start + k], gold_tokens[k])
207
- )
204
+ # Alignment-based match count (Levenshtein) as tiebreaker
205
+ window = main_tokens[start:end]
206
+ alignment = align(window, gold_tokens, False, mfn)
207
+ align_matches = sum(1 for item in alignment if isinstance(item, Match))
208
208
 
209
209
  if score > best_score:
210
210
  best = (start, end)
211
211
  best_score = score
212
- best_len_delta = len_delta
213
- best_pos_matches = pos_matches
212
+ best_waste = waste
213
+ best_align_matches = align_matches
214
214
  elif score == best_score:
215
- if best_len_delta is None or len_delta < best_len_delta:
215
+ if best_waste is None or waste < best_waste:
216
216
  best = (start, end)
217
- best_len_delta = len_delta
218
- best_pos_matches = pos_matches
219
- elif len_delta == best_len_delta:
220
- if pos_matches > best_pos_matches:
217
+ best_waste = waste
218
+ best_align_matches = align_matches
219
+ elif waste == best_waste:
220
+ if align_matches > best_align_matches:
221
221
  best = (start, end)
222
- best_pos_matches = pos_matches
223
- elif pos_matches == best_pos_matches and end > best[1]:
222
+ best_align_matches = align_matches
223
+ elif align_matches == best_align_matches and end > best[1]:
224
224
  best = (start, end)
225
225
 
226
226
  # If no tokens overlap at all, return an empty window so the caller
@@ -0,0 +1,3 @@
1
+ 0.8.2-post.12
2
+ April 15 2026
3
+ compare beter
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: batchalign
3
- Version: 0.8.2.post11
3
+ Version: 0.8.2.post12
4
4
  Summary: Python Speech Language Sample Analysis
5
5
  Author: Brian MacWhinney, Houjun Liu
6
6
  Author-email: macw@cmu.edu, houjun@cmu.edu
@@ -1,3 +0,0 @@
1
- 0.8.2-post.11
2
- April 08 2026
3
- compare beter