bantamkit 0.29.0__tar.gz → 0.29.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bantamkit-0.29.0 → bantamkit-0.29.2}/PKG-INFO +2 -2
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/__init__.py +1 -1
- bantamkit-0.29.2/src/bantamkit/docmanifest.py +99 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/docread.py +334 -25
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/evalrun.py +17 -21
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/mcpserver.py +118 -25
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/__main__.py +25 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/store.py +67 -2
- bantamkit-0.29.2/tests/data/platform-assumption-baseline.json +38 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/docread_fixtures.py +15 -1
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_docread.py +69 -3
- bantamkit-0.29.2/tests/test_docread_ceilings.py +675 -0
- bantamkit-0.29.2/tests/test_document_manifest_parity.py +401 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_eventlog.py +15 -4
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_hostinstall.py +67 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory.py +294 -1
- bantamkit-0.29.2/tests/test_platform_assumption_gate.py +268 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_served_tool_count_records.py +112 -3
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_statusline.py +38 -6
- {bantamkit-0.29.0 → bantamkit-0.29.2}/.gitignore +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/README.md +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/contracts/default.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/manifest.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/HISTORY.md +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/README.md +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/docs/architecture.md +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/docs/runbook.md +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/issues/142-settlement-timeout.md +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/patches/0009-retry-budget.patch +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/__init__.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/config.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/errors.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/posting.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/registry.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/report.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/retry.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/settle.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/src/ledger/validate.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/tests/test_posting.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/repo/tests/test_settle.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-error-contract.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-handler-map.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-patch-before-after.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-retry-attempts.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-settlement-config.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-symbol-home.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-trace-blame.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/devteam/tasks/dt-unread-key.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-in-137.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-in-359.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-in-372.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-out-11764.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-out-4137.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-large-out-8022.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-small-137.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-small-261.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/document/tasks/doc-small-388.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/fixtures/.gitkeep +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/fixtures/catalog.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/perturbations/task-completion.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/.gitkeep +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-contact.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-invoice.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-order.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-schedule.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/extract-versions.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/nav-prod-port.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/nav-release-bundle.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-audit-retention.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-cache-ttl.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-db-port.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-deploy.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-env-endpoint.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-oncall-rotation.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-oncall.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-org-quota.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/recall-owner.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-basket-total.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-cheapest.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-compare.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-gadget-value.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-stock-total.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/evals/tasks/shop-total.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/profiles/default.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/profiles/patient.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/rubrics/.gitkeep +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/rubrics/code-quality.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/rubrics/grounded-completion.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/rubrics/task-completion.yaml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/schemas/shiftwork-checkpoint.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/skills/.gitkeep +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/skills/file-graph.md +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/skills/memory.md +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/.gitkeep +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/bantamkit_read.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/bantamkit_status.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/build_identity.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/document_list.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/document_read.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/file_graph.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/memory_compact.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/memory_recall.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/memory_save.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/shiftwork_clock_in.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/shiftwork_clock_out.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/shiftwork_status.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/skill_audit.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/_assets/tools/validate_json.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/hatch_build.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/pyproject.toml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/agent.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/assets.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/budget.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/client.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/contract.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/criticreplay.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/critique.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/eventlog.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/filegraph.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/hostinstall.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/loopguard.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/mcpreport.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/__init__.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/component.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/divergence.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/memory/layers.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/pdfread.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/profile.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/shiftwork.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/skillaudit.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/statusline.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/structured.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/src/bantamkit/textutil.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/cli_exit_status_probe.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/conftest.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/bad-crc.docx +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/charref-4301-digits.html +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/charset-table.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/compression-method-9.docx +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/corrupt-deflate.docx +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/encrypted-member.docx +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/encrypted-mimetype.odt +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/eszett-cell-ref.xlsx +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/internal-dtd-entity.docx +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/rfc2231-charset.eml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/rfc822-nested-twice.eml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/unicode-digit-shared-string.xlsx +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/docread/x-uuencode.eml +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/f8404ab-perturbation-baseline.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/data/served-tool-surface.json +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/perturbation_baseline_harness.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/rbp16_effect_probe.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/rbp18_payload_probe.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_adapter.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_agent.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_amendguard.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_bantamkit_read_tool.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_budget.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_build_identity.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_client.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_compaction_corpus_survey.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_conformance.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_contract_fanout.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_criticreplay.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_critique.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_doc_commands_gate.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_document_setup.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_document_tasks.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_document_tools.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_encoding_gate.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_evalrun.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_field_program_gates.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_field_programs.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_filegraph.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_ladder_statistics.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_launcher_which.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_layers.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_loopguard.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mcp_endpoint.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mcpdrift.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mcpreport.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mcpserver.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_compact_tool.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_component.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_divergence.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_layers.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_memory_store_tripwire.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_mutmatrix.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_newline_gate.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_packaging.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_pdfread.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_pinharness_ledger.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_shiftwork.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_skillaudit.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_status_surface.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_structured.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_thread_exception_gate.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_tool_manifest.py +0 -0
- {bantamkit-0.29.0 → bantamkit-0.29.2}/tests/test_version_agreement.py +0 -0
|
@@ -29,4 +29,4 @@ from bantamkit.structured import StructuredOutputError, extract_json, structured
|
|
|
29
29
|
# file as its dynamic version source, so the wheel's metadata and the string the MCP
|
|
30
30
|
# server advertises are the same committed bytes, and neither is a function of when
|
|
31
31
|
# someone last ran `pip`.
|
|
32
|
-
__version__ = "0.29.
|
|
32
|
+
__version__ = "0.29.2"
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""One rendering of a `docread.Document` as a manifest — for every caller that has one.
|
|
2
|
+
|
|
3
|
+
Register entries (b) and (h), `docs/roadmap-toolbox.md` row 8. The entry dict
|
|
4
|
+
`{document, kind, index, part, row_count, rows, omissions}` that `contract.document_manifest`
|
|
5
|
+
takes was built by hand in `evalrun._document_tools` and again in `mcpserver.bantamkit_read`,
|
|
6
|
+
with no shared helper and no test that the two produced the same bytes for the same file. They
|
|
7
|
+
did not: the eval harness answered a zero-row part with `document_offset_past_end`, whose
|
|
8
|
+
sentence reads `numbered 0 to -1`, while the MCP server answered `"<part>" in <path> has no
|
|
9
|
+
rows`. One runtime disagreeing with itself about a file is not a two-runtime divergence and
|
|
10
|
+
costs no `docs/porting.md` row — it just had to stop.
|
|
11
|
+
|
|
12
|
+
**Why this is its own module and not a function on either caller.** `evalrun` is the
|
|
13
|
+
measurement harness and `mcpserver` is Layer 5; making either import the other to reach a
|
|
14
|
+
shared renderer would put the eval harness behind the MCP SDK's import, or the server behind
|
|
15
|
+
`yaml` and the whole eval config surface. `docread` itself is the other candidate and is the
|
|
16
|
+
better one on paper — but the entry dict is the *contract layer's* input shape, not the
|
|
17
|
+
reader's, and `docread` is deliberately ignorant of who renders it. So the adapter between the
|
|
18
|
+
two sits here: it imports `docread` for nothing at all (it reads attributes off whatever it is
|
|
19
|
+
handed) and `contract` for the sentences, which is the direction the layer rule allows.
|
|
20
|
+
|
|
21
|
+
The sentences are `contract`'s, with ONE exception this module inherited rather than chose:
|
|
22
|
+
`document_no_rows` builds its text here, exactly as `mcpserver` built it inline before. It is
|
|
23
|
+
not in `assets/contracts/default.yaml` beside every other model-facing sentence in this
|
|
24
|
+
repository, and the port spells it a third time in `runtime-ts/src/mcp/server.ts`. Moving it
|
|
25
|
+
into the contract asset is a two-runtime contract change and is registered, not done here.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
from collections.abc import Iterable
|
|
31
|
+
from typing import Any
|
|
32
|
+
|
|
33
|
+
from bantamkit.contract import document_error, document_manifest
|
|
34
|
+
|
|
35
|
+
__all__ = [
|
|
36
|
+
"document_no_rows",
|
|
37
|
+
"manifest_entries",
|
|
38
|
+
"package_entries",
|
|
39
|
+
"render_manifest",
|
|
40
|
+
]
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def manifest_entries(document: str, doc: Any) -> list[dict]:
|
|
44
|
+
"""The PART-grain entries for one document, keyed by the label the caller uses.
|
|
45
|
+
|
|
46
|
+
`document` is the caller's name for the file — a path on the MCP surface, a fixture name
|
|
47
|
+
in the eval harness — and it is the only thing about this rendering that legitimately
|
|
48
|
+
differs between the two. Everything else is a fact about the bytes.
|
|
49
|
+
"""
|
|
50
|
+
return [
|
|
51
|
+
{
|
|
52
|
+
"document": document,
|
|
53
|
+
"kind": doc.kind,
|
|
54
|
+
"index": part.index,
|
|
55
|
+
"part": part.name,
|
|
56
|
+
"row_count": part.row_count,
|
|
57
|
+
"rows": part.rows,
|
|
58
|
+
"omissions": [o.as_dict() for o in part.omissions],
|
|
59
|
+
}
|
|
60
|
+
for part in doc.parts
|
|
61
|
+
]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def package_entries(document: str, doc: Any) -> list[dict]:
|
|
65
|
+
"""The DOCUMENT-grain entries: what the container holds that belongs to no part.
|
|
66
|
+
|
|
67
|
+
Empty when the document has no such omissions, because `contract.document_manifest`
|
|
68
|
+
renders an entry without them byte-for-byte as it did before the disclosure existed —
|
|
69
|
+
which is what keeps the committed `document-read` rows where they are.
|
|
70
|
+
"""
|
|
71
|
+
if not doc.omissions:
|
|
72
|
+
return []
|
|
73
|
+
return [{"document": document, "omissions": [o.as_dict() for o in doc.omissions]}]
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def render_manifest(documents: Iterable[tuple[str, Any]]) -> str:
|
|
77
|
+
"""The manifest for one or many (label, `Document`) pairs, as one observation."""
|
|
78
|
+
parts: list[dict] = []
|
|
79
|
+
package: list[dict] = []
|
|
80
|
+
for document, doc in documents:
|
|
81
|
+
parts.extend(manifest_entries(document, doc))
|
|
82
|
+
package.extend(package_entries(document, doc))
|
|
83
|
+
return document_manifest(parts, package)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def document_no_rows(part: str, document: str) -> str:
|
|
87
|
+
"""The refusal for a part that has no rows at all — NOT an offset past the end.
|
|
88
|
+
|
|
89
|
+
`document_offset_past_end` over a zero-row part prints `numbered 0 to -1`, a range with
|
|
90
|
+
no members, and it says "offset N is past the end" about an offset of 0, which is not
|
|
91
|
+
past anything. No offset can be in range here, so the reply states that fact instead.
|
|
92
|
+
|
|
93
|
+
This exact string is pinned ON THE WIRE for both runtimes by
|
|
94
|
+
`tools/conformance/suites/wire.mjs` (`read: id 12`), which asserts it as a literal on each
|
|
95
|
+
side precisely so that both sides drifting back together would still fail. Changing it is
|
|
96
|
+
a two-runtime change plus that case; it is why this is the sentence that survived (b)/(h)
|
|
97
|
+
rather than the eval harness's.
|
|
98
|
+
"""
|
|
99
|
+
return document_error(f'"{part}" in {document} has no rows')
|
|
@@ -210,15 +210,37 @@ OMIT_UNMAPPED = "unmapped-text"
|
|
|
210
210
|
# votes on a 4096-byte head, so `extract` routinely meets bytes that verdict never saw, and
|
|
211
211
|
# those bytes can contradict it. The prefix that IS text is content; the rest is COUNTED.
|
|
212
212
|
OMIT_UNREAD_TAIL = "unread-tail"
|
|
213
|
-
#
|
|
214
|
-
#
|
|
215
|
-
#
|
|
213
|
+
# What a ceiling THIS READER imposes kept out of the rows. Not a property of the file, which
|
|
214
|
+
# is exactly why it is declared as a count instead of applied in silence. Four ceilings carry
|
|
215
|
+
# it, and each one names its own number in `what`: bytes past `TEXT_MAX_BYTES` for a plain-text
|
|
216
|
+
# file, for the markup of an `.html`/`.mhtml`, and for what `textutil` converted a `.doc` or
|
|
217
|
+
# an `.rtf` into — the same constant three times, because "how much of a file this reader
|
|
218
|
+
# reads" is one number and not one per container — and rows past `XLSX_MAX_TEXT_BYTES` for a
|
|
219
|
+
# workbook, where the thing that runs away is the RENDERING rather than the file.
|
|
220
|
+
#
|
|
221
|
+
# The principle overclaimed until review round 5 (H3), and the correction is worth stating
|
|
222
|
+
# because the sentence is what stopped anyone looking: `.doc` and `.rtf` had NO ceiling of any
|
|
223
|
+
# kind — `extract_textutil` rendered every byte of the converter's stdout — and `.docx` had
|
|
224
|
+
# none either, measured at 40,000,000 bytes of text out of a 181,289-byte file, 2.38x
|
|
225
|
+
# `TEXT_MAX_BYTES`, with both omission tuples empty.
|
|
226
|
+
#
|
|
227
|
+
# The fifth ceiling is deliberately NOT one of these, and that is the honest amendment rather
|
|
228
|
+
# than a fifth token: `ZIP_MEMBER_MAX_BYTES` refuses instead of disclosing, because half an
|
|
229
|
+
# XML member is not a smaller XML member. So an OOXML container is bounded by what this reader
|
|
230
|
+
# will PARSE and says so by refusing; every other container is bounded by what it will READ
|
|
231
|
+
# and says so by counting. `.docx` needs no rendering budget on top of that: the text a
|
|
232
|
+
# `<w:t>` walk produces can never exceed the bytes of the part it walked.
|
|
216
233
|
OMIT_SIZE_CAP = "size-cap"
|
|
217
234
|
# A worksheet cell whose own `r` reference could not place it: not letters-then-digits, or a
|
|
218
235
|
# column past the last one the format has. The cell's TEXT is in the rows, at its XML
|
|
219
236
|
# position; what the rows do not carry is the column the file asked for. Counted per reason,
|
|
220
237
|
# with the columns it landed in, exactly as `number-format` is counted per format code.
|
|
221
238
|
OMIT_UNPLACED_CELL = "unplaced-cell"
|
|
239
|
+
# A worksheet cell a LATER cell in the same row and column replaced. Last-wins is what both
|
|
240
|
+
# runtimes do and what a writer's own second cell means, so the reading is kept; what was not
|
|
241
|
+
# kept was the disclosure. Counted with the columns it happened in, like every other cell this
|
|
242
|
+
# reader could not put in the rows.
|
|
243
|
+
OMIT_DUPLICATE_CELL = "duplicate-cell"
|
|
222
244
|
|
|
223
245
|
|
|
224
246
|
class DocumentReadError(BantamError):
|
|
@@ -621,6 +643,33 @@ _UNREADABLE_OPTIONAL = (
|
|
|
621
643
|
) # fmt: skip
|
|
622
644
|
|
|
623
645
|
|
|
646
|
+
# How much ONE MEMBER of a zip this reader will decompress and parse. The ceiling on the
|
|
647
|
+
# PARSE, which is a different door from `XLSX_MAX_TEXT_BYTES`: that one bounds the text a
|
|
648
|
+
# workbook renders, and it never stands in front of this, because `_read` decompresses the
|
|
649
|
+
# member whole and `_parse` builds a tree from it before the first row is rendered.
|
|
650
|
+
#
|
|
651
|
+
# MEASURED 2026-09-06 on this machine, through `docread.extract` at the shipped constants,
|
|
652
|
+
# on a sheet of N `<row><c t="inlineStr"><is><t>x</t></is></c></row>` deflated at level 9:
|
|
653
|
+
# 400,000 rows are 58,097 bytes on disk and peaked at 342.2 MB (5,890x) with NO omission;
|
|
654
|
+
# 1,000,000 rows are 143,658 bytes and peaked at 838.8 MB (5,839x). Linear and unbounded —
|
|
655
|
+
# the memory is the `Element` tree, not the text, so a budget over the rendering could not
|
|
656
|
+
# see it. At this ceiling the same shape parses in 0.74 s and peaks at 264.3 MB, which is a
|
|
657
|
+
# worst case rather than no case at all.
|
|
658
|
+
#
|
|
659
|
+
# 16 MiB, the number `TEXT_MAX_BYTES` states, because "how much of a file this reader reads"
|
|
660
|
+
# is one number — but under its OWN NAME, because it bounds a member of a container and not a
|
|
661
|
+
# file on a disk, and a bar that varies one must not be varying the other. MEASURED the same
|
|
662
|
+
# day over `~/Downloads`, `~/Documents/Claude/Projects` and `~/Documents` (pruned as J25
|
|
663
|
+
# prunes them): 25 OOXML/ODF packages, the largest single XML member among them 4,283,286
|
|
664
|
+
# bytes — 3.9x under this — and the largest `word/document.xml` 159,976 bytes, 105x under it.
|
|
665
|
+
#
|
|
666
|
+
# A REFUSAL and not a truncation, which is the one place this module departs from
|
|
667
|
+
# "disclose, never truncate": half an XML document is not a smaller XML document, and a tree
|
|
668
|
+
# built from a severed member would carry text that is not what the file says. So the reader
|
|
669
|
+
# stops and names the member, the number the file declares and its own ceiling.
|
|
670
|
+
ZIP_MEMBER_MAX_BYTES = 16 * 1024 * 1024
|
|
671
|
+
|
|
672
|
+
|
|
624
673
|
def _read(zf: zipfile.ZipFile, name: str, path: Path) -> bytes:
|
|
625
674
|
"""One member this reader cannot do without, or a refusal in the reader's own words.
|
|
626
675
|
|
|
@@ -642,8 +691,29 @@ def _read(zf: zipfile.ZipFile, name: str, path: Path) -> bytes:
|
|
|
642
691
|
different facts about the file (a stored checksum that lies, a deflate stream that is
|
|
643
692
|
not one), and it is the only part of the sentence the Node port cannot print from the
|
|
644
693
|
same words — the ruling in docs/porting.md quotes it.
|
|
694
|
+
|
|
695
|
+
And a fifth, which is not zipfile's: a member that decompresses past
|
|
696
|
+
`ZIP_MEMBER_MAX_BYTES`. The gate is the UNCOMPRESSED SIZE the central directory declares,
|
|
697
|
+
read before anything is decompressed, and it is deliberately the cheapest possible check —
|
|
698
|
+
a member that says it is 46 MB costs no inflate at all to refuse.
|
|
699
|
+
|
|
700
|
+
Those are the attacker's bytes, so the question is what a lying declaration buys, and the
|
|
701
|
+
answer was MEASURED rather than assumed: `zipfile.ZipExtFile` clamps its own output to
|
|
702
|
+
`ZipInfo.file_size` and checks the CRC of what it produced, so a member declaring 10 bytes
|
|
703
|
+
while holding 100,000 yields 10 bytes and `BadZipFile: Bad CRC-32` — declaring LOW is a
|
|
704
|
+
damaged file, not a way past the ceiling, and declaring HIGH is what this refuses. That
|
|
705
|
+
clamp is CPython's and not the format's: the Node port walks the archive with its own
|
|
706
|
+
reader, and if that reader does not clamp it owes the property a ceiling on the real read.
|
|
707
|
+
The property is the contract; the mechanism is not.
|
|
645
708
|
"""
|
|
646
709
|
try:
|
|
710
|
+
declared = zf.getinfo(name).file_size
|
|
711
|
+
if declared > ZIP_MEMBER_MAX_BYTES:
|
|
712
|
+
raise DocumentReadError(
|
|
713
|
+
f"{path.name} is a zip but its {name} declares {declared} bytes uncompressed, "
|
|
714
|
+
f"past the {ZIP_MEMBER_MAX_BYTES} bytes this reader parses, "
|
|
715
|
+
"so this reader cannot parse it"
|
|
716
|
+
)
|
|
647
717
|
return zf.read(name)
|
|
648
718
|
except KeyError:
|
|
649
719
|
sample = ", ".join(sorted(zf.namelist())[:8]) or "(empty archive)"
|
|
@@ -866,6 +936,35 @@ _MAX_COLUMN_LETTERS = 3
|
|
|
866
936
|
# into the manifest with no bound at all.
|
|
867
937
|
UNPLACED_SHAPE = "the column of a cell whose reference is not letters then digits"
|
|
868
938
|
UNPLACED_RANGE = "the column of a cell past XFD, the last column the format has"
|
|
939
|
+
# What a cell loses when a later cell in the same row claims its column. Fixed for the same
|
|
940
|
+
# reason the two above are: the reference is the file's bytes and the column letter is this
|
|
941
|
+
# reader's own, so only the letter travels — in `where`, bounded by `XLSX_MAX_COLUMNS`.
|
|
942
|
+
DUPLICATE_CELL = "the text of a cell a later cell in the same row and column replaced"
|
|
943
|
+
|
|
944
|
+
# How much text ONE WORKBOOK may materialise, all sheets together. `XLSX_MAX_COLUMNS` bounds
|
|
945
|
+
# a ROW and nothing bounded the document, which is a ceiling with a hole in it: the width is
|
|
946
|
+
# bought a row at a time. MEASURED 2026-09-06 on both runtimes: 20,000 rows each holding one
|
|
947
|
+
# `XFD1` cell deflate to 53,967 bytes and materialise 327,680,000 bytes in 9.02 s — 6,072x,
|
|
948
|
+
# out of a file small enough to mail.
|
|
949
|
+
#
|
|
950
|
+
# 16 MiB, the same figure `TEXT_MAX_BYTES` carries and for the same kind of reason, but under
|
|
951
|
+
# its OWN NAME because the two bound different things: bytes read off a disk there, bytes
|
|
952
|
+
# rendered out of a container here, and a bar that varies one must not be varying the other.
|
|
953
|
+
# MEASURED 2026-09-06 over `~/Downloads`, `~/Documents/Claude/Projects` and `~/Documents`
|
|
954
|
+
# (pruned as J25 prunes them): 15 `.xlsx`, the largest 8,664,227 bytes on disk, and the
|
|
955
|
+
# largest RENDERING among them 754,520 bytes — 22x under this budget, so no real workbook on
|
|
956
|
+
# this machine meets it.
|
|
957
|
+
#
|
|
958
|
+
# It bounds the RENDERING and nothing else. The sentence that stood here said it bounded the
|
|
959
|
+
# rendering "because the file is already bounded", and that was FALSE for the whole life of
|
|
960
|
+
# this constant: `_read` decompressed a member whole and `_parse` built a tree from it, both
|
|
961
|
+
# before the first row was rendered and both outside this budget, so 58,097 bytes on disk
|
|
962
|
+
# peaked at 342.2 MB with no omission and this ceiling never saw it (review round 5, H1).
|
|
963
|
+
# What bounds the file is `ZIP_MEMBER_MAX_BYTES`, one door earlier; this bounds what comes out
|
|
964
|
+
# of it. The shortfall is disclosed as `OMIT_SIZE_CAP` counting the rows no sheet rendered: a
|
|
965
|
+
# row is what a caller addresses, and a byte count of text that was never built would be a
|
|
966
|
+
# number this reader cannot honestly produce.
|
|
967
|
+
XLSX_MAX_TEXT_BYTES = 16 * 1024 * 1024
|
|
869
968
|
|
|
870
969
|
|
|
871
970
|
def _column(ref: str | None, fallback: int) -> tuple[int, str]:
|
|
@@ -947,8 +1046,29 @@ def _cell_text(cell: ET.Element, shared: list[str]) -> str:
|
|
|
947
1046
|
return raw # numbers, cached formula strings (`str`), errors (`e`): stored form, verbatim
|
|
948
1047
|
|
|
949
1048
|
|
|
1049
|
+
@dataclass
|
|
1050
|
+
class _TextBudget:
|
|
1051
|
+
"""How much rendered text one WORKBOOK may still materialise, and what it cost to stop.
|
|
1052
|
+
|
|
1053
|
+
One of these is made per `extract_xlsx` call and handed to every sheet, which is the whole
|
|
1054
|
+
point: a budget made per sheet would let an N-sheet workbook materialise N budgets, and the
|
|
1055
|
+
input this ceiling exists for is one sheet of 20,000 rows anyway.
|
|
1056
|
+
|
|
1057
|
+
`total` counts every `<row>` the document declares, rendered or not, because the omission
|
|
1058
|
+
has to say what the rows it did render are a fraction OF. Counting them costs a walk of
|
|
1059
|
+
XML that is already parsed and bounded by the file; rendering them is what does not.
|
|
1060
|
+
"""
|
|
1061
|
+
|
|
1062
|
+
remaining: int
|
|
1063
|
+
total: int = 0
|
|
1064
|
+
dropped: int = 0
|
|
1065
|
+
|
|
1066
|
+
|
|
950
1067
|
def _sheet_rows(
|
|
951
|
-
root: ET.Element,
|
|
1068
|
+
root: ET.Element,
|
|
1069
|
+
shared: list[str],
|
|
1070
|
+
date_styles: tuple[str, ...] = (),
|
|
1071
|
+
budget: _TextBudget | None = None,
|
|
952
1072
|
) -> tuple[tuple[str, ...], tuple[Omission, ...]]:
|
|
953
1073
|
"""The rendered rows of one sheet, and a count of what the rendering did not carry.
|
|
954
1074
|
|
|
@@ -959,19 +1079,40 @@ def _sheet_rows(
|
|
|
959
1079
|
A cell `_column` cannot place is one of those omissions and NOT a refusal (review round 4,
|
|
960
1080
|
M2): it keeps its text at its XML position and loses only the column the file asked for.
|
|
961
1081
|
Counted per reason and rendered after the format codes, so the order of this tuple is
|
|
962
|
-
blank rows, then number formats by code, then unplaced cells by reason
|
|
1082
|
+
blank rows, then number formats by code, then unplaced cells by reason, then the cells a
|
|
1083
|
+
later cell in the same row and column replaced.
|
|
1084
|
+
|
|
1085
|
+
A DUPLICATE is a cell whose column already holds text from a cell earlier in the same row.
|
|
1086
|
+
Last-wins is kept — it is what both runtimes do and what a writer's own second cell means
|
|
1087
|
+
— and the earlier cell's text is disclosed rather than dropped in silence, which is the
|
|
1088
|
+
only part of that behaviour nobody chose.
|
|
1089
|
+
|
|
1090
|
+
`budget` is the DOCUMENT's, not this sheet's: the width of one row is already bounded by
|
|
1091
|
+
`XLSX_MAX_COLUMNS` and the height of a workbook was not, so the row is where the ceiling
|
|
1092
|
+
has to bite. A row is skipped whole rather than cut in half — half a row is a row this
|
|
1093
|
+
reader cannot vouch for, which is the same rule `extract_text` follows at its own cap —
|
|
1094
|
+
so the overshoot is at most one row, itself bounded at `XLSX_MAX_COLUMNS` fields.
|
|
963
1095
|
"""
|
|
1096
|
+
if budget is None:
|
|
1097
|
+
budget = _TextBudget(XLSX_MAX_TEXT_BYTES)
|
|
964
1098
|
rows = []
|
|
965
1099
|
blank = 0
|
|
966
1100
|
dated: dict[str, dict[int, int]] = {}
|
|
967
1101
|
unplaced: dict[str, dict[int, int]] = {}
|
|
1102
|
+
duplicated: dict[int, int] = {}
|
|
968
1103
|
for row in root.iter(NS_S + "row"):
|
|
1104
|
+
budget.total += 1
|
|
1105
|
+
if budget.remaining <= 0:
|
|
1106
|
+
budget.dropped += 1
|
|
1107
|
+
continue
|
|
969
1108
|
cells: dict[int, str] = {}
|
|
970
1109
|
for position, cell in enumerate(row.iter(NS_S + "c")):
|
|
971
1110
|
text = _clean(_cell_text(cell, shared))
|
|
972
1111
|
if not text:
|
|
973
1112
|
continue
|
|
974
1113
|
column, unplaceable = _column(cell.get("r"), position)
|
|
1114
|
+
if column in cells:
|
|
1115
|
+
duplicated[column] = duplicated.get(column, 0) + 1
|
|
975
1116
|
cells[column] = text
|
|
976
1117
|
if unplaceable:
|
|
977
1118
|
unplaced.setdefault(unplaceable, {})
|
|
@@ -986,6 +1127,7 @@ def _sheet_rows(
|
|
|
986
1127
|
dated[code][column] = dated[code].get(column, 0) + 1
|
|
987
1128
|
width = max(cells) + 1 if cells else 0
|
|
988
1129
|
line = "\t".join(cells.get(i, "") for i in range(width))
|
|
1130
|
+
budget.remaining -= len(line.encode())
|
|
989
1131
|
if not line:
|
|
990
1132
|
blank += 1
|
|
991
1133
|
rows.append(line)
|
|
@@ -1012,6 +1154,15 @@ def _sheet_rows(
|
|
|
1012
1154
|
what=reason,
|
|
1013
1155
|
)
|
|
1014
1156
|
)
|
|
1157
|
+
if duplicated:
|
|
1158
|
+
omissions.append(
|
|
1159
|
+
Omission(
|
|
1160
|
+
OMIT_DUPLICATE_CELL,
|
|
1161
|
+
sum(duplicated.values()),
|
|
1162
|
+
where=tuple(_letter(c) for c in sorted(duplicated)),
|
|
1163
|
+
what=DUPLICATE_CELL,
|
|
1164
|
+
)
|
|
1165
|
+
)
|
|
1015
1166
|
return tuple(rows), tuple(omissions)
|
|
1016
1167
|
|
|
1017
1168
|
|
|
@@ -1051,7 +1202,15 @@ def _media_omission(media: dict[str, int]) -> tuple[Omission, ...]:
|
|
|
1051
1202
|
|
|
1052
1203
|
|
|
1053
1204
|
def extract_xlsx(path: str | Path) -> Document:
|
|
1205
|
+
"""Every declared sheet, under ONE `XLSX_MAX_TEXT_BYTES` budget for the whole workbook.
|
|
1206
|
+
|
|
1207
|
+
The budget is the document's, so it is made here and not in `_sheet_rows`, and the
|
|
1208
|
+
shortfall is disclosed at the document's grain for the same reason — it is not a fact
|
|
1209
|
+
about the sheet the budget happened to run out on. The cap is stated BEFORE the media
|
|
1210
|
+
tally, because the media a reader met is a count of what it met underneath the cap.
|
|
1211
|
+
"""
|
|
1054
1212
|
path = Path(path)
|
|
1213
|
+
budget = _TextBudget(XLSX_MAX_TEXT_BYTES)
|
|
1055
1214
|
with _open(path) as zf:
|
|
1056
1215
|
shared = _shared_strings(zf, path)
|
|
1057
1216
|
date_styles = _date_formats(zf)
|
|
@@ -1061,10 +1220,20 @@ def extract_xlsx(path: str | Path) -> Document:
|
|
|
1061
1220
|
if target is None:
|
|
1062
1221
|
raise DocumentReadError(f"sheet {name!r} has no resolvable worksheet part")
|
|
1063
1222
|
sheet = _parse(_read(zf, target, path), target, path)
|
|
1064
|
-
rows, omissions = _sheet_rows(sheet, shared, date_styles)
|
|
1223
|
+
rows, omissions = _sheet_rows(sheet, shared, date_styles, budget)
|
|
1065
1224
|
omissions = _media_omission(_anchored_media(zf, target, media)) + omissions
|
|
1066
1225
|
parts.append(Part(name=name, index=index, rows=rows, omissions=omissions))
|
|
1067
|
-
|
|
1226
|
+
capped: tuple[Omission, ...] = ()
|
|
1227
|
+
if budget.dropped:
|
|
1228
|
+
capped = (
|
|
1229
|
+
Omission(
|
|
1230
|
+
OMIT_SIZE_CAP,
|
|
1231
|
+
budget.dropped,
|
|
1232
|
+
what=f"{budget.total} rows in this workbook; this reader renders "
|
|
1233
|
+
f"{XLSX_MAX_TEXT_BYTES} bytes of cell text",
|
|
1234
|
+
),
|
|
1235
|
+
)
|
|
1236
|
+
return Document(kind="xlsx", parts=tuple(parts), omissions=capped + _media_omission(media))
|
|
1068
1237
|
|
|
1069
1238
|
|
|
1070
1239
|
def extract_docx(path: str | Path) -> Document:
|
|
@@ -1136,22 +1305,50 @@ def _cap_charrefs(text: str) -> str:
|
|
|
1136
1305
|
|
|
1137
1306
|
|
|
1138
1307
|
def _with_bounded_unescape(name: str):
|
|
1139
|
-
"""`HTMLParser.<name
|
|
1308
|
+
"""`HTMLParser.<name>` with `unescape` bound to the capped one, or `None` if it cannot be.
|
|
1140
1309
|
|
|
1141
1310
|
The two methods that call `unescape` (`goahead` on text, `parse_starttag` on attribute
|
|
1142
1311
|
values) look it up in `html.parser`'s globals; a copy of the code object with one entry
|
|
1143
1312
|
of that namespace replaced is the same loop calling the same helpers, and nothing else
|
|
1144
1313
|
in the process — no other parser, not `html.unescape` itself — sees the change.
|
|
1314
|
+
|
|
1315
|
+
CONTAINED, `docs/roadmap-toolbox.md` row 8 entry (u). This depends on three properties of
|
|
1316
|
+
a CPython private method at once — the name existing, `unescape` resolving as a MODULE
|
|
1317
|
+
GLOBAL, and no closure — and it is called in a class body, so before this guard a single
|
|
1318
|
+
changed property raised `AttributeError` (or `TypeError`) at MODULE IMPORT and the MCP
|
|
1319
|
+
server did not start. `pyproject.toml` declares `requires-python = ">=3.11"` and CI
|
|
1320
|
+
measures 3.11 and 3.12, so every interpreter from 3.13 up is permitted and none is
|
|
1321
|
+
measured; an interpreter this package says it supports may not be able to make it fail
|
|
1322
|
+
to import. Measured on this machine's CPython 3.12.13: `goahead` has `'unescape' in
|
|
1323
|
+
co_names` True and `co_freevars ()`, `parse_starttag` the same.
|
|
1324
|
+
|
|
1325
|
+
All three are checked rather than caught, because the failure that is NOT an exception is
|
|
1326
|
+
the dangerous one: a method that resolves `unescape` some other way would take this
|
|
1327
|
+
rebinding silently and go on calling the uncapped `html.unescape`. `None` here is what
|
|
1328
|
+
`html_rows` reads to fall back, and `HTML_UNESCAPE_BOUNDED` is what makes that visible.
|
|
1145
1329
|
"""
|
|
1146
|
-
method = getattr(html.parser.HTMLParser, name)
|
|
1330
|
+
method = getattr(html.parser.HTMLParser, name, None)
|
|
1331
|
+
code = getattr(method, "__code__", None)
|
|
1332
|
+
if code is None or "unescape" not in code.co_names or code.co_freevars:
|
|
1333
|
+
return None
|
|
1147
1334
|
return types.FunctionType(
|
|
1148
|
-
|
|
1335
|
+
code,
|
|
1149
1336
|
{**vars(html.parser), "unescape": lambda text: html.unescape(_cap_charrefs(text))},
|
|
1150
1337
|
name,
|
|
1151
1338
|
method.__defaults__,
|
|
1152
1339
|
)
|
|
1153
1340
|
|
|
1154
1341
|
|
|
1342
|
+
# The rebindings that could be built on THIS interpreter, and whether both of them could.
|
|
1343
|
+
# `html_rows` reads the flag, a test can force it, and nothing about the fallback is silent.
|
|
1344
|
+
_BOUNDED_UNESCAPE = {
|
|
1345
|
+
name: bound
|
|
1346
|
+
for name in ("goahead", "parse_starttag")
|
|
1347
|
+
if (bound := _with_bounded_unescape(name)) is not None
|
|
1348
|
+
}
|
|
1349
|
+
HTML_UNESCAPE_BOUNDED = len(_BOUNDED_UNESCAPE) == 2
|
|
1350
|
+
|
|
1351
|
+
|
|
1155
1352
|
class _HtmlText(html.parser.HTMLParser):
|
|
1156
1353
|
"""HTML to rows. Block markup ends a row, `<td>`/`<th>` separate fields with a tab.
|
|
1157
1354
|
|
|
@@ -1171,8 +1368,9 @@ class _HtmlText(html.parser.HTMLParser):
|
|
|
1171
1368
|
# `handle_entityref` changes the library's chunking — `Ab` is handed over as `&#`
|
|
1172
1369
|
# and `65b`, and an `&#` with no `;` anywhere after it makes the parser emit the rest
|
|
1173
1370
|
# of the document, tags included, as data at `close()`.
|
|
1174
|
-
|
|
1175
|
-
|
|
1371
|
+
# The rebindings are attached AFTER the class statement, from `_BOUNDED_UNESCAPE`, so an
|
|
1372
|
+
# interpreter that has neither method still produces a class — see `_with_bounded_unescape`
|
|
1373
|
+
# and `html_rows` for what happens then.
|
|
1176
1374
|
|
|
1177
1375
|
def __init__(self) -> None:
|
|
1178
1376
|
super().__init__(convert_charrefs=True)
|
|
@@ -1218,8 +1416,22 @@ class _HtmlText(html.parser.HTMLParser):
|
|
|
1218
1416
|
self._flush()
|
|
1219
1417
|
|
|
1220
1418
|
|
|
1419
|
+
for _name, _bound in _BOUNDED_UNESCAPE.items():
|
|
1420
|
+
setattr(_HtmlText, _name, _bound)
|
|
1421
|
+
|
|
1422
|
+
|
|
1221
1423
|
def html_rows(markup: str) -> tuple[str, ...]:
|
|
1222
|
-
"""Rendered rows of one HTML fragment. A pure function of the string: no I/O, no host.
|
|
1424
|
+
"""Rendered rows of one HTML fragment. A pure function of the string: no I/O, no host.
|
|
1425
|
+
|
|
1426
|
+
When the per-chunk binding could not be built on this interpreter, the WHOLE MARKUP is
|
|
1427
|
+
capped first — the pre-H1 shape, kept as the fallback. It is measurably worse and it is
|
|
1428
|
+
measurably not a crash: H1 refused it as the default because `&#<4301 digits>;` inside
|
|
1429
|
+
`<xmp>` is CDATA the parser never unescapes, so this rewrite turns it into U+FFFD where
|
|
1430
|
+
the bounded binding keeps the digits. A reader that answers slightly differently beats a
|
|
1431
|
+
package that will not import, and the difference is one a test can see.
|
|
1432
|
+
"""
|
|
1433
|
+
if not HTML_UNESCAPE_BOUNDED:
|
|
1434
|
+
markup = _cap_charrefs(markup)
|
|
1223
1435
|
parser = _HtmlText()
|
|
1224
1436
|
parser.feed(markup)
|
|
1225
1437
|
parser.close()
|
|
@@ -1251,6 +1463,30 @@ def _decoded_body(part: email.message.Message) -> str:
|
|
|
1251
1463
|
return payload.decode("utf-8", errors="replace")
|
|
1252
1464
|
|
|
1253
1465
|
|
|
1466
|
+
def _read_to_ceiling(path: Path) -> tuple[bytes, tuple[Omission, ...]]:
|
|
1467
|
+
"""A file's bytes up to `TEXT_MAX_BYTES`, and the `OMIT_SIZE_CAP` for what is past it.
|
|
1468
|
+
|
|
1469
|
+
ONE ceiling and one sentence for every container that holds markup, because "how much of
|
|
1470
|
+
a file this reader reads" is a fact about the reader and not about the suffix: a 1 GB
|
|
1471
|
+
`.txt` stopped at `TEXT_MAX_BYTES` and counted the rest while a 1 GB `.html` was held
|
|
1472
|
+
whole (`docs/roadmap-toolbox.md` row 8, entry (k)). The sentence is `extract_text`'s,
|
|
1473
|
+
to the byte, so a caller cannot tell from it which reader hit the cap.
|
|
1474
|
+
|
|
1475
|
+
Where `extract_text` also cuts back to the last line break, this does not: markup is not
|
|
1476
|
+
a line-oriented format, a half-open tag is not a claim about content the way half a line
|
|
1477
|
+
is, and `html.parser` closes what the file left open without inventing text for it.
|
|
1478
|
+
"""
|
|
1479
|
+
size = path.stat().st_size
|
|
1480
|
+
with path.open("rb") as handle:
|
|
1481
|
+
raw = handle.read(TEXT_MAX_BYTES + 1)
|
|
1482
|
+
if len(raw) <= TEXT_MAX_BYTES:
|
|
1483
|
+
return raw, ()
|
|
1484
|
+
raw = raw[:TEXT_MAX_BYTES]
|
|
1485
|
+
dropped = size - TEXT_MAX_BYTES
|
|
1486
|
+
what = f"{size} bytes on disk; this reader reads {TEXT_MAX_BYTES}"
|
|
1487
|
+
return raw, (Omission(OMIT_SIZE_CAP, dropped, size=dropped, what=what),)
|
|
1488
|
+
|
|
1489
|
+
|
|
1254
1490
|
def extract_mhtml(path: str | Path) -> Document:
|
|
1255
1491
|
"""A MIME message / MHTML archive: every text part, in message order.
|
|
1256
1492
|
|
|
@@ -1261,8 +1497,12 @@ def extract_mhtml(path: str | Path) -> Document:
|
|
|
1261
1497
|
`-format html` pass left `signature` split as `s= ignature`.
|
|
1262
1498
|
"""
|
|
1263
1499
|
path = Path(path)
|
|
1264
|
-
|
|
1265
|
-
|
|
1500
|
+
# BOUNDED, entry (k): `message_from_binary_file` reads the handle to EOF, so a 1 GB
|
|
1501
|
+
# `.mht` was held whole — the same hole `extract_html` had, in its own spelling. The
|
|
1502
|
+
# message is parsed from the bytes this reader will admit to having read; a truncated
|
|
1503
|
+
# MIME message is one `email` still walks, and what it could not see is COUNTED.
|
|
1504
|
+
raw, capped = _read_to_ceiling(path)
|
|
1505
|
+
message = email.message_from_bytes(raw, policy=email.policy.default)
|
|
1266
1506
|
bodies: list[tuple[str, str]] = []
|
|
1267
1507
|
skipped: dict[str, int] = {}
|
|
1268
1508
|
skipped_bytes = 0
|
|
@@ -1286,7 +1526,7 @@ def extract_mhtml(path: str | Path) -> Document:
|
|
|
1286
1526
|
rows = html_rows(body) if subtype == "html" else _plain_rows(body)
|
|
1287
1527
|
name = "document" if len(bodies) == 1 else f"part{index}"
|
|
1288
1528
|
parts.append(Part(name=name, index=index, rows=rows))
|
|
1289
|
-
omissions = ()
|
|
1529
|
+
omissions: tuple[Omission, ...] = ()
|
|
1290
1530
|
if skipped:
|
|
1291
1531
|
omissions = (
|
|
1292
1532
|
Omission(
|
|
@@ -1296,18 +1536,29 @@ def extract_mhtml(path: str | Path) -> Document:
|
|
|
1296
1536
|
what=", ".join(sorted(skipped)),
|
|
1297
1537
|
),
|
|
1298
1538
|
)
|
|
1539
|
+
# The cap first: the media tally counts what this reader met UNDERNEATH it, so a caller
|
|
1540
|
+
# who reads that number without the cap above it has read a lower bound as a total.
|
|
1299
1541
|
return _nonempty(
|
|
1300
|
-
Document(kind="mhtml", parts=tuple(parts), omissions=omissions),
|
|
1542
|
+
Document(kind="mhtml", parts=tuple(parts), omissions=capped + omissions),
|
|
1301
1543
|
path,
|
|
1302
1544
|
"no text/html or text/plain part carried any text",
|
|
1303
1545
|
)
|
|
1304
1546
|
|
|
1305
1547
|
|
|
1306
1548
|
def extract_html(path: str | Path) -> Document:
|
|
1549
|
+
"""Markup to rows, up to `TEXT_MAX_BYTES` of it, saying how much it did not read.
|
|
1550
|
+
|
|
1551
|
+
BOUNDED, entry (k): this did `path.read_bytes()`, so a 1 GB `.html` was materialised whole
|
|
1552
|
+
where a 1 GB `.txt` had stopped at the cap and counted the rest since J10.
|
|
1553
|
+
"""
|
|
1307
1554
|
path = Path(path)
|
|
1308
|
-
raw = path
|
|
1555
|
+
raw, capped = _read_to_ceiling(path)
|
|
1309
1556
|
markup = raw.decode("utf-8", errors="replace")
|
|
1310
|
-
doc = Document(
|
|
1557
|
+
doc = Document(
|
|
1558
|
+
kind="html",
|
|
1559
|
+
parts=(Part(name="document", index=0, rows=html_rows(markup)),),
|
|
1560
|
+
omissions=capped,
|
|
1561
|
+
)
|
|
1311
1562
|
return _nonempty(doc, path, "its markup carried no text outside script and style")
|
|
1312
1563
|
|
|
1313
1564
|
|
|
@@ -1321,13 +1572,45 @@ def _nonempty(doc: Document, path: Path, why: str) -> Document:
|
|
|
1321
1572
|
`.xlsx` is deliberately exempt: a declared-but-empty sheet is a real part with no rows and
|
|
1322
1573
|
J10's row counts are committed measurements. Here there is no such thing — a `.doc` that
|
|
1323
1574
|
renders nothing is a `.doc` this reader did not read.
|
|
1575
|
+
|
|
1576
|
+
THE REFUSAL CARRIES THE OMISSIONS, review round 5 (H2). `extract_html` and `extract_mhtml`
|
|
1577
|
+
build the `Document` with `capped` in `omissions` and hand it here, and here it raises: the
|
|
1578
|
+
refusal kept the reader's verdict about the content and threw away the ceiling that
|
|
1579
|
+
produced that verdict. MEASURED on a 16,777,291-byte `.html` whose 16 MiB `<script>`
|
|
1580
|
+
comment is followed by one visible sentence — `its markup carried no text outside script
|
|
1581
|
+
and style ... it is not an empty document`, about a document that carries text, from a
|
|
1582
|
+
read that stopped 75 bytes short of it. The same on a `.mht` past the ceiling, where the
|
|
1583
|
+
media tally went with it.
|
|
1584
|
+
|
|
1585
|
+
A reader is allowed to refuse. It is not allowed to state a false fact about a file, and
|
|
1586
|
+
`why` is a fact about the file only when the whole file was read. Under a cap it is scoped
|
|
1587
|
+
to the part that was read and the unread bytes are named, so a caller who would have
|
|
1588
|
+
stopped looking has the one number that tells it not to.
|
|
1324
1589
|
"""
|
|
1325
1590
|
if any(part.rows for part in doc.parts):
|
|
1326
1591
|
return doc
|
|
1327
|
-
|
|
1328
|
-
|
|
1329
|
-
|
|
1330
|
-
|
|
1592
|
+
capped = next((o for o in doc.omissions if o.subject == OMIT_SIZE_CAP), None)
|
|
1593
|
+
media = next((o for o in doc.omissions if o.subject == OMIT_MEDIA), None)
|
|
1594
|
+
if capped is None:
|
|
1595
|
+
said = (
|
|
1596
|
+
f"cannot read {path.name}: it is a {doc.kind} container but {why}, so this reader "
|
|
1597
|
+
"has no text for it — it is not an empty document"
|
|
1598
|
+
)
|
|
1599
|
+
else:
|
|
1600
|
+
# The cap's own `what` is quoted rather than rebuilt, so the sentence and the omission
|
|
1601
|
+
# can never state different numbers, and so the clause says whose bytes were counted:
|
|
1602
|
+
# a file's on disk here, `textutil`'s output there.
|
|
1603
|
+
said = (
|
|
1604
|
+
f"cannot read {path.name}: it is a {doc.kind} container but {why} in the part "
|
|
1605
|
+
f"this reader read ({capped.what}) — the {capped.count} bytes it did not read "
|
|
1606
|
+
"may carry text"
|
|
1607
|
+
)
|
|
1608
|
+
if media is not None:
|
|
1609
|
+
said += (
|
|
1610
|
+
f", and it holds {media.count} embedded part(s) ({media.what}) this reader "
|
|
1611
|
+
"renders no text for"
|
|
1612
|
+
)
|
|
1613
|
+
raise DocumentReadError(said)
|
|
1331
1614
|
|
|
1332
1615
|
|
|
1333
1616
|
# --------------------------------------------------------------- plain text, in no container
|
|
@@ -1492,6 +1775,29 @@ def _textutil_type(exe: str, path: Path) -> str:
|
|
|
1492
1775
|
return ""
|
|
1493
1776
|
|
|
1494
1777
|
|
|
1778
|
+
def _cap_converted(stdout: bytes) -> tuple[bytes, tuple[Omission, ...]]:
|
|
1779
|
+
"""`textutil`'s output up to `TEXT_MAX_BYTES`, and the `OMIT_SIZE_CAP` for what is past it.
|
|
1780
|
+
|
|
1781
|
+
THE FOURTH CEILING, review round 5 (H3). `.doc` and `.rtf` had none at all: every byte the
|
|
1782
|
+
converter wrote was rendered, so the principle stated at `OMIT_SIZE_CAP` — "how much of a
|
|
1783
|
+
file this reader reads" is one number and not one per container — was contradicted two
|
|
1784
|
+
containers over by the same module.
|
|
1785
|
+
|
|
1786
|
+
The number is `TEXT_MAX_BYTES`, and `what` names WHOSE bytes were counted rather than
|
|
1787
|
+
borrowing `_read_to_ceiling`'s sentence: these are the converter's, not the file's, and a
|
|
1788
|
+
caller must not read this omission as a statement about the `.rtf` on disk. What is NOT
|
|
1789
|
+
bounded here is the memory: `_textutil_run` captures the whole of a subprocess's stdout
|
|
1790
|
+
before this sees a byte of it, so this bounds what the reader RENDERS and the host process
|
|
1791
|
+
still decides how much it wrote. Saying so is the point — a comment that claimed otherwise
|
|
1792
|
+
is exactly what (v) shipped.
|
|
1793
|
+
"""
|
|
1794
|
+
if len(stdout) <= TEXT_MAX_BYTES:
|
|
1795
|
+
return stdout, ()
|
|
1796
|
+
dropped = len(stdout) - TEXT_MAX_BYTES
|
|
1797
|
+
what = f"{len(stdout)} bytes {TEXTUTIL} produced; this reader reads {TEXT_MAX_BYTES}"
|
|
1798
|
+
return stdout[:TEXT_MAX_BYTES], (Omission(OMIT_SIZE_CAP, dropped, size=dropped, what=what),)
|
|
1799
|
+
|
|
1800
|
+
|
|
1495
1801
|
def extract_textutil(path: str | Path, kind: str = "doc") -> Document:
|
|
1496
1802
|
"""Convert through `textutil` and render its plain text as rows.
|
|
1497
1803
|
|
|
@@ -1519,8 +1825,11 @@ def extract_textutil(path: str | Path, kind: str = "doc") -> Document:
|
|
|
1519
1825
|
"raw bytes re-encoded, not the document's text"
|
|
1520
1826
|
)
|
|
1521
1827
|
done = _textutil_run(exe, path, "-convert", "txt", "-stdout")
|
|
1522
|
-
|
|
1523
|
-
|
|
1828
|
+
produced, capped = _cap_converted(done.stdout)
|
|
1829
|
+
rows = _plain_rows(produced.decode("utf-8", errors="replace"))
|
|
1830
|
+
doc = Document(
|
|
1831
|
+
kind=kind, parts=(Part(name="document", index=0, rows=rows),), omissions=capped
|
|
1832
|
+
)
|
|
1524
1833
|
return _nonempty(doc, path, f"{TEXTUTIL} converted it to no text at all")
|
|
1525
1834
|
|
|
1526
1835
|
|