tensorless 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tensorless-0.2.0/PKG-INFO +13 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/architecture.md +2 -1
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/configuration.md +2 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/installation.md +0 -20
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/limitations.md +4 -6
- {tensorless-0.1.0 → tensorless-0.2.0}/examples/text_generation_example.py +6 -6
- {tensorless-0.1.0 → tensorless-0.2.0}/pyproject.toml +1 -1
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/auto/config.py +5 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/config.py +4 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/runtime.py +2 -2
- tensorless-0.2.0/tensorless/tokenization/__init__.py +11 -0
- tensorless-0.2.0/tensorless/tokenization/bpe_tokenizer.py +123 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/tokenization/char_tokenizer.py +1 -1
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/training/data_prep.py +11 -4
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/training/trainer.py +2 -2
- tensorless-0.2.0/tensorless.egg-info/PKG-INFO +13 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless.egg-info/SOURCES.txt +1 -1
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_train_text_generation.py +18 -0
- tensorless-0.1.0/PKG-INFO +0 -111
- tensorless-0.1.0/README.md +0 -97
- tensorless-0.1.0/tensorless/tokenization/__init__.py +0 -3
- tensorless-0.1.0/tensorless.egg-info/PKG-INFO +0 -111
- {tensorless-0.1.0 → tensorless-0.2.0}/LICENSE +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/MANIFEST.in +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/api_reference.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/automatic_mode.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/checkpointing.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/cli.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/contributing.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/examples.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/inference.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/quickstart.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/roadmap.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/tl_format.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/training.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/troubleshooting.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/docs/tutorial.md +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/examples/tabular_classification_example.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/examples/tabular_regression_example.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/examples/text_classification_example.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/setup.cfg +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/__init__.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/_version.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/api.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/auto/__init__.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/auto/detector.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/checkpoint/__init__.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/checkpoint/manager.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/cli/__init__.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/cli/main.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/data/__init__.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/data/fingerprint.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/data/inspector.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/data/loader.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/data/tabular.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/devices/__init__.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/devices/device.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/errors.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/models/__init__.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/models/mlp.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/models/registry.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/models/transformer.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/serialization/__init__.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/serialization/tl_format.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/training/__init__.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless/training/early_stopping.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless.egg-info/dependency_links.txt +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless.egg-info/entry_points.txt +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless.egg-info/requires.txt +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tensorless.egg-info/top_level.txt +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_auto_detection.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_checkpoint_resume.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_cli.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_data_loading.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_end_to_end.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_fingerprint.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_serialization.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_train_tabular.py +0 -0
- {tensorless-0.1.0 → tensorless-0.2.0}/tests/test_train_text_classification.py +0 -0
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tensorless
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: ML with maximum automation and minimum setup.
|
|
5
|
+
Author: Tensorless Contributors
|
|
6
|
+
License: MIT
|
|
7
|
+
Requires-Python: >=3.9
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: torch>=2.0
|
|
11
|
+
Provides-Extra: dev
|
|
12
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
13
|
+
Dynamic: license-file
|
|
@@ -22,7 +22,8 @@ tensorless/
|
|
|
22
22
|
│ ├── detector.py Dataset -> task string
|
|
23
23
|
│ └── config.py Dataset + TrainConfig -> ResolvedConfig
|
|
24
24
|
├── tokenization/
|
|
25
|
-
│
|
|
25
|
+
│ ├── char_tokenizer.py CharTokenizer: vocab build/encode/decode/save/load
|
|
26
|
+
│ └── bpe_tokenizer.py BPETokenizer: corpus-trained subword encoding
|
|
26
27
|
├── models/
|
|
27
28
|
│ ├── transformer.py TinyTransformer (GPT-style decoder)
|
|
28
29
|
│ ├── mlp.py TabularMLP
|
|
@@ -26,6 +26,8 @@ defaults.
|
|
|
26
26
|
| `ff_mult` | auto | Feed-forward expansion multiplier (transformer only) |
|
|
27
27
|
| `dropout` | `0.1` | Dropout probability |
|
|
28
28
|
| `max_seq_len` | `256` (text) / `1` (tabular) | Max sequence length in tokens (text tasks) |
|
|
29
|
+
| `tokenizer` | `"bpe"` | Text tokenizer: `"bpe"` or `"char"` |
|
|
30
|
+
| `bpe_vocab_size` | `1000` | Maximum vocabulary size when `tokenizer="bpe"` |
|
|
29
31
|
|
|
30
32
|
## Optimization
|
|
31
33
|
|
|
@@ -7,26 +7,6 @@
|
|
|
7
7
|
- Optional: a CUDA-capable GPU, or a TPU with `torch_xla` installed, for
|
|
8
8
|
faster training. Tensorless works fine on CPU-only machines too.
|
|
9
9
|
|
|
10
|
-
## Install from source
|
|
11
|
-
|
|
12
|
-
Tensorless isn't published to PyPI yet. Install it directly from a
|
|
13
|
-
checkout of this repository:
|
|
14
|
-
|
|
15
|
-
```bash
|
|
16
|
-
git clone https://github.com/tensorless/tensorless.git
|
|
17
|
-
cd tensorless
|
|
18
|
-
pip install -e .
|
|
19
|
-
```
|
|
20
|
-
|
|
21
|
-
The `-e` (editable) flag means changes to the source are picked up
|
|
22
|
-
immediately without reinstalling — useful if you're also contributing.
|
|
23
|
-
|
|
24
|
-
If you plan to run the test suite, install the dev extras too:
|
|
25
|
-
|
|
26
|
-
```bash
|
|
27
|
-
pip install -e ".[dev]"
|
|
28
|
-
```
|
|
29
|
-
|
|
30
10
|
## Verify your installation
|
|
31
11
|
|
|
32
12
|
```bash
|
|
@@ -6,12 +6,10 @@ current limits:
|
|
|
6
6
|
|
|
7
7
|
## Models
|
|
8
8
|
|
|
9
|
-
- The default text tokenizer is
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
tokenizers
|
|
13
|
-
more sequence positions to represent, and generation quality per
|
|
14
|
-
parameter is lower than a comparably-sized BPE-tokenized model.
|
|
9
|
+
- The default text tokenizer is the dependency-free **BPE** tokenizer, which
|
|
10
|
+
learns merges from the training corpus. It is more token-efficient than the
|
|
11
|
+
optional `tokenizer="char"` alternative, but still simpler than production
|
|
12
|
+
tokenizers such as GPT-style BPE or SentencePiece.
|
|
15
13
|
- Models are trained **from scratch** every time — there's no
|
|
16
14
|
fine-tuning of pretrained checkpoints. This keeps things dependency-
|
|
17
15
|
free and fast to set up, but means text-generation quality on small
|
|
@@ -32,12 +32,12 @@ def main():
|
|
|
32
32
|
model = tl.train(
|
|
33
33
|
CORPUS_PATH,
|
|
34
34
|
out=os.path.join(DATA_DIR, "text_gen_example.tl"),
|
|
35
|
-
epochs=5,
|
|
36
|
-
d_model=64,
|
|
37
|
-
layers=2,
|
|
38
|
-
heads=4,
|
|
39
|
-
batch_size=16,
|
|
40
|
-
max_seq_len=64,
|
|
35
|
+
# epochs=5,
|
|
36
|
+
# d_model=64,
|
|
37
|
+
# layers=2,
|
|
38
|
+
# heads=4,
|
|
39
|
+
# batch_size=16,
|
|
40
|
+
# max_seq_len=64,
|
|
41
41
|
)
|
|
42
42
|
|
|
43
43
|
print("\nGenerated text:")
|
|
@@ -71,6 +71,9 @@ def resolve_config(ds: Dataset, user: TrainConfig) -> ResolvedConfig:
|
|
|
71
71
|
model_type = user.model_type or (
|
|
72
72
|
"transformer" if task in ("text-generation", "text-classification") else "mlp"
|
|
73
73
|
)
|
|
74
|
+
tokenizer = user.tokenizer or "bpe"
|
|
75
|
+
if tokenizer not in ("char", "bpe"):
|
|
76
|
+
raise ValueError("tokenizer must be 'char' or 'bpe'")
|
|
74
77
|
|
|
75
78
|
d_model, layers, heads, ff_mult = _auto_model_size(n, ds.kind)
|
|
76
79
|
|
|
@@ -92,6 +95,8 @@ def resolve_config(ds: Dataset, user: TrainConfig) -> ResolvedConfig:
|
|
|
92
95
|
ff_mult=user.ff_mult or ff_mult,
|
|
93
96
|
dropout=user.dropout if user.dropout is not None else 0.1,
|
|
94
97
|
max_seq_len=user.max_seq_len or (256 if ds.kind in ("text", "text_labeled") else 1),
|
|
98
|
+
tokenizer=tokenizer,
|
|
99
|
+
bpe_vocab_size=user.bpe_vocab_size or 1000,
|
|
95
100
|
optimizer=user.optimizer or "adamw",
|
|
96
101
|
learning_rate=user.learning_rate or (3e-4 if model_type == "transformer" else 1e-3),
|
|
97
102
|
weight_decay=user.weight_decay if user.weight_decay is not None else 0.01,
|
|
@@ -33,6 +33,8 @@ class TrainConfig:
|
|
|
33
33
|
ff_mult: Optional[int] = None
|
|
34
34
|
dropout: Optional[float] = None
|
|
35
35
|
max_seq_len: Optional[int] = None
|
|
36
|
+
tokenizer: Optional[str] = None # "char" or "bpe"
|
|
37
|
+
bpe_vocab_size: Optional[int] = None
|
|
36
38
|
|
|
37
39
|
# --- optimization ---
|
|
38
40
|
optimizer: Optional[str] = None # "adamw", "adam", "sgd"
|
|
@@ -89,6 +91,8 @@ class ResolvedConfig:
|
|
|
89
91
|
ff_mult: int
|
|
90
92
|
dropout: float
|
|
91
93
|
max_seq_len: int
|
|
94
|
+
tokenizer: str
|
|
95
|
+
bpe_vocab_size: int
|
|
92
96
|
|
|
93
97
|
optimizer: str
|
|
94
98
|
learning_rate: float
|
|
@@ -16,7 +16,7 @@ from typing import Any, Dict, List, Union
|
|
|
16
16
|
import torch
|
|
17
17
|
|
|
18
18
|
from .models.registry import build_model
|
|
19
|
-
from .tokenization
|
|
19
|
+
from .tokenization import tokenizer_from_state_dict
|
|
20
20
|
from .data.tabular import TabularPreprocessor
|
|
21
21
|
from .devices.device import get_torch_device
|
|
22
22
|
from .errors import ModelError
|
|
@@ -35,7 +35,7 @@ class LoadedModel:
|
|
|
35
35
|
|
|
36
36
|
self.tokenizer = None
|
|
37
37
|
if payload.get("tokenizer_state") is not None:
|
|
38
|
-
self.tokenizer =
|
|
38
|
+
self.tokenizer = tokenizer_from_state_dict(payload["tokenizer_state"])
|
|
39
39
|
|
|
40
40
|
self.preprocessor = None
|
|
41
41
|
if payload.get("preprocessor_state") is not None:
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
from .bpe_tokenizer import BPETokenizer
|
|
2
|
+
from .char_tokenizer import CharTokenizer
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def tokenizer_from_state_dict(state):
|
|
6
|
+
if state.get("tokenizer_type", "char") == "bpe":
|
|
7
|
+
return BPETokenizer.from_state_dict(state)
|
|
8
|
+
return CharTokenizer.from_state_dict(state)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
__all__ = ["CharTokenizer", "BPETokenizer", "tokenizer_from_state_dict"]
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""A small, dependency-free byte-pair encoding tokenizer."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import Counter
|
|
6
|
+
from typing import Dict, List, Sequence, Tuple
|
|
7
|
+
|
|
8
|
+
from .char_tokenizer import BOS, EOS, PAD, SPECIAL_TOKENS, UNK
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class BPETokenizer:
|
|
12
|
+
"""Character-initialized BPE tokenizer suitable for small local corpora.
|
|
13
|
+
|
|
14
|
+
BPE operates on Unicode characters rather than bytes so the resulting
|
|
15
|
+
vocabulary remains portable and decoding preserves the original text.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
def __init__(self, vocab: List[str], merges: List[Tuple[str, str]]):
|
|
19
|
+
self.vocab = list(vocab)
|
|
20
|
+
self.merges = [tuple(pair) for pair in merges]
|
|
21
|
+
self._stoi: Dict[str, int] = {token: i for i, token in enumerate(self.vocab)}
|
|
22
|
+
|
|
23
|
+
@property
|
|
24
|
+
def vocab_size(self) -> int:
|
|
25
|
+
return len(self.vocab)
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def pad_id(self) -> int:
|
|
29
|
+
return self._stoi[PAD]
|
|
30
|
+
|
|
31
|
+
@property
|
|
32
|
+
def unk_id(self) -> int:
|
|
33
|
+
return self._stoi[UNK]
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def bos_id(self) -> int:
|
|
37
|
+
return self._stoi[BOS]
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def eos_id(self) -> int:
|
|
41
|
+
return self._stoi[EOS]
|
|
42
|
+
|
|
43
|
+
@classmethod
|
|
44
|
+
def build(cls, texts: Sequence[str], vocab_size: int = 1000) -> "BPETokenizer":
|
|
45
|
+
if vocab_size < len(SPECIAL_TOKENS):
|
|
46
|
+
raise ValueError(f"vocab_size must be at least {len(SPECIAL_TOKENS)}")
|
|
47
|
+
|
|
48
|
+
symbols = set(char for text in texts for char in text)
|
|
49
|
+
vocab = list(SPECIAL_TOKENS) + sorted(symbols - set(SPECIAL_TOKENS))
|
|
50
|
+
sequences = [list(text) for text in texts]
|
|
51
|
+
merges: List[Tuple[str, str]] = []
|
|
52
|
+
|
|
53
|
+
while len(vocab) < vocab_size:
|
|
54
|
+
counts = Counter(
|
|
55
|
+
(sequence[index], sequence[index + 1])
|
|
56
|
+
for sequence in sequences
|
|
57
|
+
for index in range(len(sequence) - 1)
|
|
58
|
+
)
|
|
59
|
+
if not counts:
|
|
60
|
+
break
|
|
61
|
+
best_pair, best_count = min(
|
|
62
|
+
counts.items(), key=lambda item: (-item[1], item[0])
|
|
63
|
+
)
|
|
64
|
+
if best_count < 2:
|
|
65
|
+
break
|
|
66
|
+
merged = best_pair[0] + best_pair[1]
|
|
67
|
+
if merged in vocab:
|
|
68
|
+
break
|
|
69
|
+
merges.append(best_pair)
|
|
70
|
+
vocab.append(merged)
|
|
71
|
+
sequences = [cls._apply_merge(sequence, best_pair, merged) for sequence in sequences]
|
|
72
|
+
|
|
73
|
+
return cls(vocab, merges)
|
|
74
|
+
|
|
75
|
+
@staticmethod
|
|
76
|
+
def _apply_merge(sequence: List[str], pair: Tuple[str, str], merged: str) -> List[str]:
|
|
77
|
+
result: List[str] = []
|
|
78
|
+
index = 0
|
|
79
|
+
while index < len(sequence):
|
|
80
|
+
if index + 1 < len(sequence) and (sequence[index], sequence[index + 1]) == pair:
|
|
81
|
+
result.append(merged)
|
|
82
|
+
index += 2
|
|
83
|
+
else:
|
|
84
|
+
result.append(sequence[index])
|
|
85
|
+
index += 1
|
|
86
|
+
return result
|
|
87
|
+
|
|
88
|
+
def _tokenize(self, text: str) -> List[str]:
|
|
89
|
+
tokens = list(text)
|
|
90
|
+
for pair in self.merges:
|
|
91
|
+
tokens = self._apply_merge(tokens, pair, pair[0] + pair[1])
|
|
92
|
+
return tokens
|
|
93
|
+
|
|
94
|
+
def encode(self, text: str, add_special_tokens: bool = True) -> List[int]:
|
|
95
|
+
ids = [self._stoi.get(token, self.unk_id) for token in self._tokenize(text)]
|
|
96
|
+
if add_special_tokens:
|
|
97
|
+
ids = [self.bos_id] + ids + [self.eos_id]
|
|
98
|
+
return ids
|
|
99
|
+
|
|
100
|
+
def decode(self, ids: Sequence[int], skip_special_tokens: bool = True) -> str:
|
|
101
|
+
special_ids = {self.pad_id, self.bos_id, self.eos_id} if skip_special_tokens else set()
|
|
102
|
+
result = []
|
|
103
|
+
for index in ids:
|
|
104
|
+
if index in special_ids:
|
|
105
|
+
continue
|
|
106
|
+
if 0 <= index < len(self.vocab):
|
|
107
|
+
token = self.vocab[index]
|
|
108
|
+
if token == UNK:
|
|
109
|
+
result.append("\ufffd")
|
|
110
|
+
elif token not in SPECIAL_TOKENS:
|
|
111
|
+
result.append(token)
|
|
112
|
+
return "".join(result)
|
|
113
|
+
|
|
114
|
+
def state_dict(self) -> Dict:
|
|
115
|
+
return {
|
|
116
|
+
"tokenizer_type": "bpe",
|
|
117
|
+
"vocab": self.vocab,
|
|
118
|
+
"merges": [list(pair) for pair in self.merges],
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
@classmethod
|
|
122
|
+
def from_state_dict(cls, state: Dict) -> "BPETokenizer":
|
|
123
|
+
return cls(list(state["vocab"]), [tuple(pair) for pair in state.get("merges", [])])
|
|
@@ -16,6 +16,7 @@ from torch.utils.data import DataLoader, Dataset as TorchDataset
|
|
|
16
16
|
|
|
17
17
|
from ..data.loader import Dataset
|
|
18
18
|
from ..data.tabular import TabularPreprocessor
|
|
19
|
+
from ..tokenization.bpe_tokenizer import BPETokenizer
|
|
19
20
|
from ..tokenization.char_tokenizer import CharTokenizer
|
|
20
21
|
from ..auto.detector import target_column
|
|
21
22
|
from ..errors import DataError
|
|
@@ -87,10 +88,16 @@ def _split_indices(n: int, val_split: float, seed: int) -> Tuple[List[int], List
|
|
|
87
88
|
return train_idx, val_idx
|
|
88
89
|
|
|
89
90
|
|
|
91
|
+
def _build_tokenizer(ds: Dataset, cfg: Dict[str, Any]):
|
|
92
|
+
if cfg.get("tokenizer", "bpe") == "bpe":
|
|
93
|
+
return BPETokenizer.build(ds.texts, vocab_size=cfg.get("bpe_vocab_size", 1000))
|
|
94
|
+
return CharTokenizer.build(ds.texts)
|
|
95
|
+
|
|
96
|
+
|
|
90
97
|
def prepare_text_generation(
|
|
91
|
-
ds: Dataset, cfg: Dict[str, Any], tokenizer
|
|
98
|
+
ds: Dataset, cfg: Dict[str, Any], tokenizer=None
|
|
92
99
|
) -> PreparedData:
|
|
93
|
-
tokenizer = tokenizer or
|
|
100
|
+
tokenizer = tokenizer or _build_tokenizer(ds, cfg)
|
|
94
101
|
all_ids: List[int] = []
|
|
95
102
|
for t in ds.texts:
|
|
96
103
|
all_ids.extend(tokenizer.encode(t, add_special_tokens=True))
|
|
@@ -129,10 +136,10 @@ def prepare_text_generation(
|
|
|
129
136
|
def prepare_text_classification(
|
|
130
137
|
ds: Dataset,
|
|
131
138
|
cfg: Dict[str, Any],
|
|
132
|
-
tokenizer
|
|
139
|
+
tokenizer=None,
|
|
133
140
|
classes: Optional[List[str]] = None,
|
|
134
141
|
) -> PreparedData:
|
|
135
|
-
tokenizer = tokenizer or
|
|
142
|
+
tokenizer = tokenizer or _build_tokenizer(ds, cfg)
|
|
136
143
|
classes = classes or sorted(set(ds.labels))
|
|
137
144
|
label2id = {c: i for i, c in enumerate(classes)}
|
|
138
145
|
|
|
@@ -19,7 +19,7 @@ from ..data.loader import Dataset
|
|
|
19
19
|
from ..devices.device import get_torch_device
|
|
20
20
|
from ..checkpoint.manager import CheckpointManager
|
|
21
21
|
from ..models.registry import build_model
|
|
22
|
-
from ..tokenization
|
|
22
|
+
from ..tokenization import tokenizer_from_state_dict
|
|
23
23
|
from ..data.tabular import TabularPreprocessor
|
|
24
24
|
from .early_stopping import EarlyStopping
|
|
25
25
|
from . import data_prep as dp
|
|
@@ -97,7 +97,7 @@ def run_training(
|
|
|
97
97
|
preprocessor = None
|
|
98
98
|
if resume_state is not None:
|
|
99
99
|
if resume_state.get("tokenizer_state") is not None:
|
|
100
|
-
tokenizer =
|
|
100
|
+
tokenizer = tokenizer_from_state_dict(resume_state["tokenizer_state"])
|
|
101
101
|
if resume_state.get("preprocessor_state") is not None:
|
|
102
102
|
preprocessor = TabularPreprocessor.from_state_dict(resume_state["preprocessor_state"])
|
|
103
103
|
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tensorless
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: ML with maximum automation and minimum setup.
|
|
5
|
+
Author: Tensorless Contributors
|
|
6
|
+
License: MIT
|
|
7
|
+
Requires-Python: >=3.9
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: torch>=2.0
|
|
11
|
+
Provides-Extra: dev
|
|
12
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
13
|
+
Dynamic: license-file
|
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
LICENSE
|
|
2
2
|
MANIFEST.in
|
|
3
|
-
README.md
|
|
4
3
|
pyproject.toml
|
|
5
4
|
docs/api_reference.md
|
|
6
5
|
docs/architecture.md
|
|
@@ -56,6 +55,7 @@ tensorless/models/transformer.py
|
|
|
56
55
|
tensorless/serialization/__init__.py
|
|
57
56
|
tensorless/serialization/tl_format.py
|
|
58
57
|
tensorless/tokenization/__init__.py
|
|
58
|
+
tensorless/tokenization/bpe_tokenizer.py
|
|
59
59
|
tensorless/tokenization/char_tokenizer.py
|
|
60
60
|
tensorless/training/__init__.py
|
|
61
61
|
tensorless/training/data_prep.py
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import os
|
|
2
2
|
|
|
3
3
|
import tensorless as tl
|
|
4
|
+
from tensorless.tokenization.bpe_tokenizer import BPETokenizer
|
|
4
5
|
from tensorless.serialization.tl_format import load_tl
|
|
5
6
|
|
|
6
7
|
from .conftest import TINY_TEXT_KWARGS
|
|
@@ -10,6 +11,7 @@ def test_train_text_generation_creates_tl_file(text_corpus, workdir):
|
|
|
10
11
|
model = tl.train(text_corpus, out="model.tl", **TINY_TEXT_KWARGS)
|
|
11
12
|
assert os.path.isfile("model.tl")
|
|
12
13
|
assert model.task == "text-generation"
|
|
14
|
+
assert isinstance(model.tokenizer, BPETokenizer)
|
|
13
15
|
assert model.info()["training_complete"] is True
|
|
14
16
|
|
|
15
17
|
|
|
@@ -21,6 +23,22 @@ def test_generate_after_reload(text_corpus, workdir):
|
|
|
21
23
|
assert len(text) > 0
|
|
22
24
|
|
|
23
25
|
|
|
26
|
+
def test_bpe_tokenizer_trains_and_reloads(text_corpus, workdir):
|
|
27
|
+
model = tl.train(
|
|
28
|
+
text_corpus,
|
|
29
|
+
out="bpe.tl",
|
|
30
|
+
tokenizer="bpe",
|
|
31
|
+
bpe_vocab_size=64,
|
|
32
|
+
**TINY_TEXT_KWARGS,
|
|
33
|
+
)
|
|
34
|
+
assert isinstance(model.tokenizer, BPETokenizer)
|
|
35
|
+
assert model.tokenizer.decode(model.tokenizer.encode("the quick")) == "the quick"
|
|
36
|
+
|
|
37
|
+
reloaded = tl.load("bpe.tl")
|
|
38
|
+
assert isinstance(reloaded.tokenizer, BPETokenizer)
|
|
39
|
+
assert reloaded.tokenizer.decode(reloaded.tokenizer.encode("the quick")) == "the quick"
|
|
40
|
+
|
|
41
|
+
|
|
24
42
|
def test_cpu_training_works(text_corpus, workdir):
|
|
25
43
|
model = tl.train(text_corpus, out="model.tl", device="cpu", **TINY_TEXT_KWARGS)
|
|
26
44
|
assert model.config["device"] == "cpu"
|
tensorless-0.1.0/PKG-INFO
DELETED
|
@@ -1,111 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: tensorless
|
|
3
|
-
Version: 0.1.0
|
|
4
|
-
Summary: ML with maximum automation and minimum setup.
|
|
5
|
-
Author: Tensorless Contributors
|
|
6
|
-
License: MIT
|
|
7
|
-
Requires-Python: >=3.9
|
|
8
|
-
Description-Content-Type: text/markdown
|
|
9
|
-
License-File: LICENSE
|
|
10
|
-
Requires-Dist: torch>=2.0
|
|
11
|
-
Provides-Extra: dev
|
|
12
|
-
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
13
|
-
Dynamic: license-file
|
|
14
|
-
|
|
15
|
-
# Tensorless
|
|
16
|
-
|
|
17
|
-
**ML with maximum automation and minimum setup.**
|
|
18
|
-
|
|
19
|
-
```python
|
|
20
|
-
import tensorless as tl
|
|
21
|
-
|
|
22
|
-
tl.train("./data")
|
|
23
|
-
```
|
|
24
|
-
|
|
25
|
-
That's it. Tensorless inspects your dataset, figures out what kind of
|
|
26
|
-
task you're trying to solve, builds and configures a model, trains it,
|
|
27
|
-
validates it, checkpoints it, and saves a single portable `model.tl`
|
|
28
|
-
file you can move anywhere.
|
|
29
|
-
|
|
30
|
-
```python
|
|
31
|
-
model = tl.run("model.tl") # interactive chat, if it's a text model
|
|
32
|
-
# or
|
|
33
|
-
model = tl.load("model.tl")
|
|
34
|
-
model.predict(...)
|
|
35
|
-
```
|
|
36
|
-
|
|
37
|
-
Simple by default. Powerful when you need it:
|
|
38
|
-
|
|
39
|
-
```python
|
|
40
|
-
tl.train(
|
|
41
|
-
"./data",
|
|
42
|
-
d_model=512,
|
|
43
|
-
layers=6,
|
|
44
|
-
learning_rate=3e-4,
|
|
45
|
-
batch_size=32,
|
|
46
|
-
)
|
|
47
|
-
```
|
|
48
|
-
|
|
49
|
-
## Why Tensorless
|
|
50
|
-
|
|
51
|
-
Most ML frameworks assume you already know what model you want, how big
|
|
52
|
-
it should be, which optimizer and learning rate to use, and how to wire
|
|
53
|
-
up checkpointing and resumption yourself. Tensorless flips that: it
|
|
54
|
-
makes a reasonable, working choice for all of that automatically, and
|
|
55
|
-
lets you override exactly the parts you care about.
|
|
56
|
-
|
|
57
|
-
It also remembers what it already did. Run `tl.train("./data")` twice on
|
|
58
|
-
the same dataset and it won't retrain — it'll just hand you back the
|
|
59
|
-
model it already trained. Change the data, and it retrains. Get
|
|
60
|
-
interrupted partway through a long run, and the next call resumes right
|
|
61
|
-
where it left off. This is the **Smart Auto Check**, and it's the core
|
|
62
|
-
idea the whole framework is built around.
|
|
63
|
-
|
|
64
|
-
## Install
|
|
65
|
-
|
|
66
|
-
```bash
|
|
67
|
-
pip install -e .
|
|
68
|
-
```
|
|
69
|
-
|
|
70
|
-
See [docs/installation.md](docs/installation.md) for details and
|
|
71
|
-
requirements.
|
|
72
|
-
|
|
73
|
-
## Documentation
|
|
74
|
-
|
|
75
|
-
| Doc | What's in it |
|
|
76
|
-
|---|---|
|
|
77
|
-
| [Installation](docs/installation.md) | Requirements, install steps, verifying your setup |
|
|
78
|
-
| [Quick Start](docs/quickstart.md) | The fastest path to a trained model |
|
|
79
|
-
| [Beginner Tutorial](docs/tutorial.md) | A guided, from-scratch walkthrough |
|
|
80
|
-
| [Automatic Mode](docs/automatic_mode.md) | How auto-detection and auto-configuration work, and the Smart Auto Check |
|
|
81
|
-
| [Training](docs/training.md) | `tl.train()` in depth, all supported tasks and data formats |
|
|
82
|
-
| [Inference](docs/inference.md) | `tl.run()`, `tl.load()`, and the prediction API |
|
|
83
|
-
| [Checkpointing & Resume](docs/checkpointing.md) | How checkpoints work and how resumption is decided |
|
|
84
|
-
| [The `.tl` Format](docs/tl_format.md) | What's inside a `.tl` file and why it's portable |
|
|
85
|
-
| [Configuration](docs/configuration.md) | Every override you can pass, and what it does |
|
|
86
|
-
| [CLI](docs/cli.md) | `tensorless train / run / inspect / info` |
|
|
87
|
-
| [API Reference](docs/api_reference.md) | Full function/class signatures |
|
|
88
|
-
| [Examples](docs/examples.md) | Worked examples for each supported task |
|
|
89
|
-
| [Troubleshooting](docs/troubleshooting.md) | Common errors and what to do about them |
|
|
90
|
-
| [Architecture](docs/architecture.md) | How the codebase is organized, for contributors |
|
|
91
|
-
| [Contributing](docs/contributing.md) | How to add models, backends, or data formats |
|
|
92
|
-
| [Roadmap](docs/roadmap.md) | What's planned |
|
|
93
|
-
| [Limitations](docs/limitations.md) | What Tensorless deliberately doesn't do (yet) |
|
|
94
|
-
|
|
95
|
-
## Supported today
|
|
96
|
-
|
|
97
|
-
- **Text generation** (language modeling) from `.txt`/`.md` files or JSON/JSONL with a `text` field
|
|
98
|
-
- **Text classification** from a directory of class subfolders (`positive/`, `negative/`, ...) or labeled JSON/JSONL
|
|
99
|
-
- **Tabular classification and regression** from CSV/TSV/JSON/JSONL with a target column
|
|
100
|
-
|
|
101
|
-
## Project status
|
|
102
|
-
|
|
103
|
-
Tensorless is an early-stage, actively developed framework. The core
|
|
104
|
-
loop — inspect, auto-configure, train, checkpoint, save, reload, infer —
|
|
105
|
-
is real and tested end-to-end (see [tests/](tests/)). See
|
|
106
|
-
[docs/limitations.md](docs/limitations.md) for what's intentionally out
|
|
107
|
-
of scope right now, and [docs/roadmap.md](docs/roadmap.md) for what's next.
|
|
108
|
-
|
|
109
|
-
## License
|
|
110
|
-
|
|
111
|
-
MIT
|
tensorless-0.1.0/README.md
DELETED
|
@@ -1,97 +0,0 @@
|
|
|
1
|
-
# Tensorless
|
|
2
|
-
|
|
3
|
-
**ML with maximum automation and minimum setup.**
|
|
4
|
-
|
|
5
|
-
```python
|
|
6
|
-
import tensorless as tl
|
|
7
|
-
|
|
8
|
-
tl.train("./data")
|
|
9
|
-
```
|
|
10
|
-
|
|
11
|
-
That's it. Tensorless inspects your dataset, figures out what kind of
|
|
12
|
-
task you're trying to solve, builds and configures a model, trains it,
|
|
13
|
-
validates it, checkpoints it, and saves a single portable `model.tl`
|
|
14
|
-
file you can move anywhere.
|
|
15
|
-
|
|
16
|
-
```python
|
|
17
|
-
model = tl.run("model.tl") # interactive chat, if it's a text model
|
|
18
|
-
# or
|
|
19
|
-
model = tl.load("model.tl")
|
|
20
|
-
model.predict(...)
|
|
21
|
-
```
|
|
22
|
-
|
|
23
|
-
Simple by default. Powerful when you need it:
|
|
24
|
-
|
|
25
|
-
```python
|
|
26
|
-
tl.train(
|
|
27
|
-
"./data",
|
|
28
|
-
d_model=512,
|
|
29
|
-
layers=6,
|
|
30
|
-
learning_rate=3e-4,
|
|
31
|
-
batch_size=32,
|
|
32
|
-
)
|
|
33
|
-
```
|
|
34
|
-
|
|
35
|
-
## Why Tensorless
|
|
36
|
-
|
|
37
|
-
Most ML frameworks assume you already know what model you want, how big
|
|
38
|
-
it should be, which optimizer and learning rate to use, and how to wire
|
|
39
|
-
up checkpointing and resumption yourself. Tensorless flips that: it
|
|
40
|
-
makes a reasonable, working choice for all of that automatically, and
|
|
41
|
-
lets you override exactly the parts you care about.
|
|
42
|
-
|
|
43
|
-
It also remembers what it already did. Run `tl.train("./data")` twice on
|
|
44
|
-
the same dataset and it won't retrain — it'll just hand you back the
|
|
45
|
-
model it already trained. Change the data, and it retrains. Get
|
|
46
|
-
interrupted partway through a long run, and the next call resumes right
|
|
47
|
-
where it left off. This is the **Smart Auto Check**, and it's the core
|
|
48
|
-
idea the whole framework is built around.
|
|
49
|
-
|
|
50
|
-
## Install
|
|
51
|
-
|
|
52
|
-
```bash
|
|
53
|
-
pip install -e .
|
|
54
|
-
```
|
|
55
|
-
|
|
56
|
-
See [docs/installation.md](docs/installation.md) for details and
|
|
57
|
-
requirements.
|
|
58
|
-
|
|
59
|
-
## Documentation
|
|
60
|
-
|
|
61
|
-
| Doc | What's in it |
|
|
62
|
-
|---|---|
|
|
63
|
-
| [Installation](docs/installation.md) | Requirements, install steps, verifying your setup |
|
|
64
|
-
| [Quick Start](docs/quickstart.md) | The fastest path to a trained model |
|
|
65
|
-
| [Beginner Tutorial](docs/tutorial.md) | A guided, from-scratch walkthrough |
|
|
66
|
-
| [Automatic Mode](docs/automatic_mode.md) | How auto-detection and auto-configuration work, and the Smart Auto Check |
|
|
67
|
-
| [Training](docs/training.md) | `tl.train()` in depth, all supported tasks and data formats |
|
|
68
|
-
| [Inference](docs/inference.md) | `tl.run()`, `tl.load()`, and the prediction API |
|
|
69
|
-
| [Checkpointing & Resume](docs/checkpointing.md) | How checkpoints work and how resumption is decided |
|
|
70
|
-
| [The `.tl` Format](docs/tl_format.md) | What's inside a `.tl` file and why it's portable |
|
|
71
|
-
| [Configuration](docs/configuration.md) | Every override you can pass, and what it does |
|
|
72
|
-
| [CLI](docs/cli.md) | `tensorless train / run / inspect / info` |
|
|
73
|
-
| [API Reference](docs/api_reference.md) | Full function/class signatures |
|
|
74
|
-
| [Examples](docs/examples.md) | Worked examples for each supported task |
|
|
75
|
-
| [Troubleshooting](docs/troubleshooting.md) | Common errors and what to do about them |
|
|
76
|
-
| [Architecture](docs/architecture.md) | How the codebase is organized, for contributors |
|
|
77
|
-
| [Contributing](docs/contributing.md) | How to add models, backends, or data formats |
|
|
78
|
-
| [Roadmap](docs/roadmap.md) | What's planned |
|
|
79
|
-
| [Limitations](docs/limitations.md) | What Tensorless deliberately doesn't do (yet) |
|
|
80
|
-
|
|
81
|
-
## Supported today
|
|
82
|
-
|
|
83
|
-
- **Text generation** (language modeling) from `.txt`/`.md` files or JSON/JSONL with a `text` field
|
|
84
|
-
- **Text classification** from a directory of class subfolders (`positive/`, `negative/`, ...) or labeled JSON/JSONL
|
|
85
|
-
- **Tabular classification and regression** from CSV/TSV/JSON/JSONL with a target column
|
|
86
|
-
|
|
87
|
-
## Project status
|
|
88
|
-
|
|
89
|
-
Tensorless is an early-stage, actively developed framework. The core
|
|
90
|
-
loop — inspect, auto-configure, train, checkpoint, save, reload, infer —
|
|
91
|
-
is real and tested end-to-end (see [tests/](tests/)). See
|
|
92
|
-
[docs/limitations.md](docs/limitations.md) for what's intentionally out
|
|
93
|
-
of scope right now, and [docs/roadmap.md](docs/roadmap.md) for what's next.
|
|
94
|
-
|
|
95
|
-
## License
|
|
96
|
-
|
|
97
|
-
MIT
|
|
@@ -1,111 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: tensorless
|
|
3
|
-
Version: 0.1.0
|
|
4
|
-
Summary: ML with maximum automation and minimum setup.
|
|
5
|
-
Author: Tensorless Contributors
|
|
6
|
-
License: MIT
|
|
7
|
-
Requires-Python: >=3.9
|
|
8
|
-
Description-Content-Type: text/markdown
|
|
9
|
-
License-File: LICENSE
|
|
10
|
-
Requires-Dist: torch>=2.0
|
|
11
|
-
Provides-Extra: dev
|
|
12
|
-
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
13
|
-
Dynamic: license-file
|
|
14
|
-
|
|
15
|
-
# Tensorless
|
|
16
|
-
|
|
17
|
-
**ML with maximum automation and minimum setup.**
|
|
18
|
-
|
|
19
|
-
```python
|
|
20
|
-
import tensorless as tl
|
|
21
|
-
|
|
22
|
-
tl.train("./data")
|
|
23
|
-
```
|
|
24
|
-
|
|
25
|
-
That's it. Tensorless inspects your dataset, figures out what kind of
|
|
26
|
-
task you're trying to solve, builds and configures a model, trains it,
|
|
27
|
-
validates it, checkpoints it, and saves a single portable `model.tl`
|
|
28
|
-
file you can move anywhere.
|
|
29
|
-
|
|
30
|
-
```python
|
|
31
|
-
model = tl.run("model.tl") # interactive chat, if it's a text model
|
|
32
|
-
# or
|
|
33
|
-
model = tl.load("model.tl")
|
|
34
|
-
model.predict(...)
|
|
35
|
-
```
|
|
36
|
-
|
|
37
|
-
Simple by default. Powerful when you need it:
|
|
38
|
-
|
|
39
|
-
```python
|
|
40
|
-
tl.train(
|
|
41
|
-
"./data",
|
|
42
|
-
d_model=512,
|
|
43
|
-
layers=6,
|
|
44
|
-
learning_rate=3e-4,
|
|
45
|
-
batch_size=32,
|
|
46
|
-
)
|
|
47
|
-
```
|
|
48
|
-
|
|
49
|
-
## Why Tensorless
|
|
50
|
-
|
|
51
|
-
Most ML frameworks assume you already know what model you want, how big
|
|
52
|
-
it should be, which optimizer and learning rate to use, and how to wire
|
|
53
|
-
up checkpointing and resumption yourself. Tensorless flips that: it
|
|
54
|
-
makes a reasonable, working choice for all of that automatically, and
|
|
55
|
-
lets you override exactly the parts you care about.
|
|
56
|
-
|
|
57
|
-
It also remembers what it already did. Run `tl.train("./data")` twice on
|
|
58
|
-
the same dataset and it won't retrain — it'll just hand you back the
|
|
59
|
-
model it already trained. Change the data, and it retrains. Get
|
|
60
|
-
interrupted partway through a long run, and the next call resumes right
|
|
61
|
-
where it left off. This is the **Smart Auto Check**, and it's the core
|
|
62
|
-
idea the whole framework is built around.
|
|
63
|
-
|
|
64
|
-
## Install
|
|
65
|
-
|
|
66
|
-
```bash
|
|
67
|
-
pip install -e .
|
|
68
|
-
```
|
|
69
|
-
|
|
70
|
-
See [docs/installation.md](docs/installation.md) for details and
|
|
71
|
-
requirements.
|
|
72
|
-
|
|
73
|
-
## Documentation
|
|
74
|
-
|
|
75
|
-
| Doc | What's in it |
|
|
76
|
-
|---|---|
|
|
77
|
-
| [Installation](docs/installation.md) | Requirements, install steps, verifying your setup |
|
|
78
|
-
| [Quick Start](docs/quickstart.md) | The fastest path to a trained model |
|
|
79
|
-
| [Beginner Tutorial](docs/tutorial.md) | A guided, from-scratch walkthrough |
|
|
80
|
-
| [Automatic Mode](docs/automatic_mode.md) | How auto-detection and auto-configuration work, and the Smart Auto Check |
|
|
81
|
-
| [Training](docs/training.md) | `tl.train()` in depth, all supported tasks and data formats |
|
|
82
|
-
| [Inference](docs/inference.md) | `tl.run()`, `tl.load()`, and the prediction API |
|
|
83
|
-
| [Checkpointing & Resume](docs/checkpointing.md) | How checkpoints work and how resumption is decided |
|
|
84
|
-
| [The `.tl` Format](docs/tl_format.md) | What's inside a `.tl` file and why it's portable |
|
|
85
|
-
| [Configuration](docs/configuration.md) | Every override you can pass, and what it does |
|
|
86
|
-
| [CLI](docs/cli.md) | `tensorless train / run / inspect / info` |
|
|
87
|
-
| [API Reference](docs/api_reference.md) | Full function/class signatures |
|
|
88
|
-
| [Examples](docs/examples.md) | Worked examples for each supported task |
|
|
89
|
-
| [Troubleshooting](docs/troubleshooting.md) | Common errors and what to do about them |
|
|
90
|
-
| [Architecture](docs/architecture.md) | How the codebase is organized, for contributors |
|
|
91
|
-
| [Contributing](docs/contributing.md) | How to add models, backends, or data formats |
|
|
92
|
-
| [Roadmap](docs/roadmap.md) | What's planned |
|
|
93
|
-
| [Limitations](docs/limitations.md) | What Tensorless deliberately doesn't do (yet) |
|
|
94
|
-
|
|
95
|
-
## Supported today
|
|
96
|
-
|
|
97
|
-
- **Text generation** (language modeling) from `.txt`/`.md` files or JSON/JSONL with a `text` field
|
|
98
|
-
- **Text classification** from a directory of class subfolders (`positive/`, `negative/`, ...) or labeled JSON/JSONL
|
|
99
|
-
- **Tabular classification and regression** from CSV/TSV/JSON/JSONL with a target column
|
|
100
|
-
|
|
101
|
-
## Project status
|
|
102
|
-
|
|
103
|
-
Tensorless is an early-stage, actively developed framework. The core
|
|
104
|
-
loop — inspect, auto-configure, train, checkpoint, save, reload, infer —
|
|
105
|
-
is real and tested end-to-end (see [tests/](tests/)). See
|
|
106
|
-
[docs/limitations.md](docs/limitations.md) for what's intentionally out
|
|
107
|
-
of scope right now, and [docs/roadmap.md](docs/roadmap.md) for what's next.
|
|
108
|
-
|
|
109
|
-
## License
|
|
110
|
-
|
|
111
|
-
MIT
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|