tensorflow-uit2721dlca 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tensorflow_uit2721dlca-0.1.0/PKG-INFO +60 -0
- tensorflow_uit2721dlca-0.1.0/README.md +49 -0
- tensorflow_uit2721dlca-0.1.0/pyproject.toml +20 -0
- tensorflow_uit2721dlca-0.1.0/setup.cfg +4 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca/__init__.py +1 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca/ex1.py +68 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca/ex2.py +71 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca/ex3.py +73 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca/ex4.py +199 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca/ex5.py +70 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca/prerequisites.py +87 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca.egg-info/PKG-INFO +60 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca.egg-info/SOURCES.txt +14 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca.egg-info/dependency_links.txt +1 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca.egg-info/requires.txt +4 -0
- tensorflow_uit2721dlca-0.1.0/src/tensorflow_uit2721dlca.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tensorflow-uit2721dlca
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Deep Learning assignment
|
|
5
|
+
Requires-Python: <3.11,>=3.9
|
|
6
|
+
Description-Content-Type: text/markdown
|
|
7
|
+
Requires-Dist: tensorflow<2.10.0,>=2.9.0
|
|
8
|
+
Requires-Dist: numpy<1.22.0,>=1.21.6
|
|
9
|
+
Requires-Dist: matplotlib<3.6.0,>=3.5.2
|
|
10
|
+
Requires-Dist: scikit-learn<2.0.0,>=1.0.0
|
|
11
|
+
|
|
12
|
+
# tensorflow-uit2721dlca
|
|
13
|
+
|
|
14
|
+
Deep Learning assignment package — UIT2721DLCA.
|
|
15
|
+
|
|
16
|
+
## Requirements
|
|
17
|
+
|
|
18
|
+
- Python 3.9 or 3.10 (TensorFlow 2.9 does not support Python 3.11+)
|
|
19
|
+
|
|
20
|
+
## Installation
|
|
21
|
+
|
|
22
|
+
`ash
|
|
23
|
+
pip install tensorflow-uit2721dlca
|
|
24
|
+
`
|
|
25
|
+
|
|
26
|
+
## Usage
|
|
27
|
+
|
|
28
|
+
Each module is a self-contained assignment exercise. Importing a module will
|
|
29
|
+
execute its code (download data, train model, display plots).
|
|
30
|
+
|
|
31
|
+
`python
|
|
32
|
+
# Run exercise 1 — VGG16 image classification on Imagenette
|
|
33
|
+
import tensorflow_uit2721dlca.ex1
|
|
34
|
+
|
|
35
|
+
# Run exercise 2 — VGG16 feature maps and filters
|
|
36
|
+
import tensorflow_uit2721dlca.ex2
|
|
37
|
+
|
|
38
|
+
# Run exercise 3 — Transfer learning / fine-tuning
|
|
39
|
+
import tensorflow_uit2721dlca.ex3
|
|
40
|
+
|
|
41
|
+
# Run exercise 4 — Image captioning with CNN + LSTM
|
|
42
|
+
import tensorflow_uit2721dlca.ex4
|
|
43
|
+
|
|
44
|
+
# Run exercise 5 — Conditional GAN (text-to-image)
|
|
45
|
+
import tensorflow_uit2721dlca.ex5
|
|
46
|
+
|
|
47
|
+
# Run the prerequisites / environment check
|
|
48
|
+
import tensorflow_uit2721dlca.prerequisites
|
|
49
|
+
`
|
|
50
|
+
|
|
51
|
+
> **Note:** The exercises download the Imagenette dataset (~100 MB) on first
|
|
52
|
+
> run via f.keras.utils.get_file. The file is cached in ~/.keras/datasets/
|
|
53
|
+
> and is not re-downloaded on subsequent runs.
|
|
54
|
+
|
|
55
|
+
## Package name vs import name
|
|
56
|
+
|
|
57
|
+
| Purpose | Name |
|
|
58
|
+
|---|---|
|
|
59
|
+
| pip install | ensorflow-uit2721dlca |
|
|
60
|
+
| Python import | ensorflow_uit2721dlca |
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# tensorflow-uit2721dlca
|
|
2
|
+
|
|
3
|
+
Deep Learning assignment package — UIT2721DLCA.
|
|
4
|
+
|
|
5
|
+
## Requirements
|
|
6
|
+
|
|
7
|
+
- Python 3.9 or 3.10 (TensorFlow 2.9 does not support Python 3.11+)
|
|
8
|
+
|
|
9
|
+
## Installation
|
|
10
|
+
|
|
11
|
+
`ash
|
|
12
|
+
pip install tensorflow-uit2721dlca
|
|
13
|
+
`
|
|
14
|
+
|
|
15
|
+
## Usage
|
|
16
|
+
|
|
17
|
+
Each module is a self-contained assignment exercise. Importing a module will
|
|
18
|
+
execute its code (download data, train model, display plots).
|
|
19
|
+
|
|
20
|
+
`python
|
|
21
|
+
# Run exercise 1 — VGG16 image classification on Imagenette
|
|
22
|
+
import tensorflow_uit2721dlca.ex1
|
|
23
|
+
|
|
24
|
+
# Run exercise 2 — VGG16 feature maps and filters
|
|
25
|
+
import tensorflow_uit2721dlca.ex2
|
|
26
|
+
|
|
27
|
+
# Run exercise 3 — Transfer learning / fine-tuning
|
|
28
|
+
import tensorflow_uit2721dlca.ex3
|
|
29
|
+
|
|
30
|
+
# Run exercise 4 — Image captioning with CNN + LSTM
|
|
31
|
+
import tensorflow_uit2721dlca.ex4
|
|
32
|
+
|
|
33
|
+
# Run exercise 5 — Conditional GAN (text-to-image)
|
|
34
|
+
import tensorflow_uit2721dlca.ex5
|
|
35
|
+
|
|
36
|
+
# Run the prerequisites / environment check
|
|
37
|
+
import tensorflow_uit2721dlca.prerequisites
|
|
38
|
+
`
|
|
39
|
+
|
|
40
|
+
> **Note:** The exercises download the Imagenette dataset (~100 MB) on first
|
|
41
|
+
> run via f.keras.utils.get_file. The file is cached in ~/.keras/datasets/
|
|
42
|
+
> and is not re-downloaded on subsequent runs.
|
|
43
|
+
|
|
44
|
+
## Package name vs import name
|
|
45
|
+
|
|
46
|
+
| Purpose | Name |
|
|
47
|
+
|---|---|
|
|
48
|
+
| pip install | ensorflow-uit2721dlca |
|
|
49
|
+
| Python import | ensorflow_uit2721dlca |
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "tensorflow-uit2721dlca"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Deep Learning assignment"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
|
|
11
|
+
requires-python = ">=3.9,<3.11"
|
|
12
|
+
dependencies = [
|
|
13
|
+
"tensorflow >=2.9.0,<2.10.0",
|
|
14
|
+
"numpy >=1.21.6,<1.22.0",
|
|
15
|
+
"matplotlib >=3.5.2,<3.6.0",
|
|
16
|
+
"scikit-learn >=1.0.0,<2.0.0",
|
|
17
|
+
]
|
|
18
|
+
|
|
19
|
+
[tool.setuptools.packages.find]
|
|
20
|
+
where = ["src"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import tensorflow as tf
|
|
2
|
+
import matplotlib.pyplot as plt
|
|
3
|
+
|
|
4
|
+
from tensorflow.keras.applications import VGG16
|
|
5
|
+
from tensorflow.keras.applications.vgg16 import preprocess_input, decode_predictions
|
|
6
|
+
|
|
7
|
+
# Download Imagenette
|
|
8
|
+
url = "https://s3.amazonaws.com/fast-ai-imageclas/imagenette2-160.tgz"
|
|
9
|
+
|
|
10
|
+
archive = tf.keras.utils.get_file(
|
|
11
|
+
"imagenette2-160.tgz",
|
|
12
|
+
url,
|
|
13
|
+
untar=True
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
# Load validation images
|
|
17
|
+
dataset = tf.keras.utils.image_dataset_from_directory(
|
|
18
|
+
archive + "/imagenette2-160/val",
|
|
19
|
+
image_size=(224, 224),
|
|
20
|
+
batch_size=5,
|
|
21
|
+
shuffle=True
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
print("Classes:", dataset.class_names)
|
|
25
|
+
|
|
26
|
+
# Get 5 images
|
|
27
|
+
images, labels = next(iter(dataset))
|
|
28
|
+
|
|
29
|
+
print("Original shape:", images.shape)
|
|
30
|
+
print("Preprocessed shape: (5, 224, 224, 3)")
|
|
31
|
+
|
|
32
|
+
# Display images and original labels
|
|
33
|
+
plt.figure(figsize=(12, 5))
|
|
34
|
+
|
|
35
|
+
for i in range(5):
|
|
36
|
+
plt.subplot(1, 5, i + 1)
|
|
37
|
+
plt.imshow(images[i].numpy().astype("uint8"))
|
|
38
|
+
plt.title(dataset.class_names[labels[i]])
|
|
39
|
+
plt.axis("off")
|
|
40
|
+
|
|
41
|
+
plt.show()
|
|
42
|
+
|
|
43
|
+
# Load pretrained VGG16
|
|
44
|
+
model = VGG16(weights="imagenet")
|
|
45
|
+
|
|
46
|
+
# Preprocess images
|
|
47
|
+
x = preprocess_input(images)
|
|
48
|
+
|
|
49
|
+
# Show preprocessed representation
|
|
50
|
+
print("Pixel value before preprocessing:", images[0][0][0])
|
|
51
|
+
print("Pixel value after preprocessing :", x[0][0][0])
|
|
52
|
+
|
|
53
|
+
# Predict
|
|
54
|
+
predictions = model.predict(x)
|
|
55
|
+
|
|
56
|
+
# Top 5 predictions
|
|
57
|
+
results = decode_predictions(predictions, top=5)
|
|
58
|
+
|
|
59
|
+
# Display predictions and compare with original labels
|
|
60
|
+
for i in range(5):
|
|
61
|
+
print("\nImage", i + 1)
|
|
62
|
+
print("Original label:", dataset.class_names[labels[i]])
|
|
63
|
+
print("Top-5 predictions:")
|
|
64
|
+
|
|
65
|
+
for _, label, probability in results[i]:
|
|
66
|
+
print(label, ":", round(probability * 100, 2), "%")
|
|
67
|
+
|
|
68
|
+
print("Predicted label:", results[i][0][1])
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import tensorflow as tf
|
|
2
|
+
import matplotlib.pyplot as plt
|
|
3
|
+
from tensorflow.keras.applications import VGG16
|
|
4
|
+
from tensorflow.keras.applications.vgg16 import preprocess_input
|
|
5
|
+
|
|
6
|
+
# Download Imagenette
|
|
7
|
+
url = "https://s3.amazonaws.com/fast-ai-imageclas/imagenette2-160.tgz"
|
|
8
|
+
|
|
9
|
+
archive = tf.keras.utils.get_file(
|
|
10
|
+
"imagenette2-160.tgz",
|
|
11
|
+
url,
|
|
12
|
+
untar=True
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
# Load validation images
|
|
16
|
+
dataset = tf.keras.utils.image_dataset_from_directory(
|
|
17
|
+
archive + "/imagenette2-160/val",
|
|
18
|
+
image_size=(224, 224),
|
|
19
|
+
batch_size=5,
|
|
20
|
+
shuffle=True
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
images, labels = next(iter(dataset))
|
|
24
|
+
|
|
25
|
+
# Load VGG16
|
|
26
|
+
model = VGG16(weights="imagenet")
|
|
27
|
+
|
|
28
|
+
# Take one image
|
|
29
|
+
image = images[0:1]
|
|
30
|
+
x = preprocess_input(image)
|
|
31
|
+
|
|
32
|
+
# Select 3 layers
|
|
33
|
+
layers = ["block1_conv1", "block2_conv2", "block5_conv3"]
|
|
34
|
+
|
|
35
|
+
# Create model to get feature maps
|
|
36
|
+
feature_model = tf.keras.Model(
|
|
37
|
+
inputs=model.input,
|
|
38
|
+
outputs=[model.get_layer(layer).output for layer in layers]
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
features = feature_model.predict(x)
|
|
42
|
+
|
|
43
|
+
# Display feature maps
|
|
44
|
+
for layer, feature in zip(layers, features):
|
|
45
|
+
|
|
46
|
+
print(layer, "shape:", feature.shape)
|
|
47
|
+
|
|
48
|
+
plt.figure(figsize=(10, 5))
|
|
49
|
+
|
|
50
|
+
for i in range(6):
|
|
51
|
+
plt.subplot(2, 3, i + 1)
|
|
52
|
+
plt.imshow(feature[0, :, :, i], cmap="viridis")
|
|
53
|
+
plt.axis("off")
|
|
54
|
+
|
|
55
|
+
plt.suptitle(layer)
|
|
56
|
+
plt.show()
|
|
57
|
+
|
|
58
|
+
# Display filters from first convolutional layer
|
|
59
|
+
filters = model.get_layer("block1_conv1").get_weights()[0]
|
|
60
|
+
|
|
61
|
+
print("Filter shape:", filters.shape)
|
|
62
|
+
|
|
63
|
+
plt.figure(figsize=(10, 5))
|
|
64
|
+
|
|
65
|
+
for i in range(6):
|
|
66
|
+
plt.subplot(2, 3, i + 1)
|
|
67
|
+
plt.imshow(filters[:, :, :, i])
|
|
68
|
+
plt.axis("off")
|
|
69
|
+
|
|
70
|
+
plt.suptitle("VGG16 Convolutional Filters")
|
|
71
|
+
plt.show()
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
import tensorflow as tf
|
|
2
|
+
import numpy as np
|
|
3
|
+
|
|
4
|
+
from tensorflow.keras.applications import VGG16
|
|
5
|
+
from tensorflow.keras.applications.vgg16 import preprocess_input
|
|
6
|
+
from tensorflow.keras.layers import GlobalAveragePooling2D, Dense
|
|
7
|
+
from tensorflow.keras.models import Sequential
|
|
8
|
+
from sklearn.metrics import confusion_matrix, classification_report
|
|
9
|
+
|
|
10
|
+
# Download Imagenette
|
|
11
|
+
url = "https://s3.amazonaws.com/fast-ai-imageclas/imagenette2-160.tgz"
|
|
12
|
+
|
|
13
|
+
archive = tf.keras.utils.get_file(
|
|
14
|
+
"imagenette2-160.tgz",
|
|
15
|
+
url,
|
|
16
|
+
untar=True
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
# Load only a small dataset
|
|
20
|
+
train = tf.keras.utils.image_dataset_from_directory(
|
|
21
|
+
archive + "/imagenette2-160/train",
|
|
22
|
+
image_size=(224, 224),
|
|
23
|
+
batch_size=20,
|
|
24
|
+
shuffle=True
|
|
25
|
+
).take(10)
|
|
26
|
+
|
|
27
|
+
test = tf.keras.utils.image_dataset_from_directory(
|
|
28
|
+
archive + "/imagenette2-160/val",
|
|
29
|
+
image_size=(224, 224),
|
|
30
|
+
batch_size=20,
|
|
31
|
+
shuffle=True
|
|
32
|
+
).take(3)
|
|
33
|
+
|
|
34
|
+
# Preprocess
|
|
35
|
+
train = train.map(lambda x, y: (preprocess_input(x), y))
|
|
36
|
+
test = test.map(lambda x, y: (preprocess_input(x), y))
|
|
37
|
+
|
|
38
|
+
# VGG16 feature extractor
|
|
39
|
+
base = VGG16(weights="imagenet", include_top=False)
|
|
40
|
+
print("Feature map shape:", base.output.shape)
|
|
41
|
+
base.trainable = False
|
|
42
|
+
|
|
43
|
+
# Classifier
|
|
44
|
+
model = Sequential([
|
|
45
|
+
base,
|
|
46
|
+
GlobalAveragePooling2D(),
|
|
47
|
+
Dense(10, activation="softmax")
|
|
48
|
+
])
|
|
49
|
+
print("Feature vector size: 512")
|
|
50
|
+
|
|
51
|
+
model.compile(
|
|
52
|
+
optimizer="adam",
|
|
53
|
+
loss="sparse_categorical_crossentropy",
|
|
54
|
+
metrics=["accuracy"]
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
# Train
|
|
58
|
+
model.fit(train, epochs=1)
|
|
59
|
+
|
|
60
|
+
# Test
|
|
61
|
+
y_true = []
|
|
62
|
+
y_pred = []
|
|
63
|
+
|
|
64
|
+
for x, y in test:
|
|
65
|
+
pred = model.predict(x, verbose=0)
|
|
66
|
+
y_true.extend(y.numpy())
|
|
67
|
+
y_pred.extend(np.argmax(pred, axis=1))
|
|
68
|
+
|
|
69
|
+
print("Confusion Matrix:")
|
|
70
|
+
print(confusion_matrix(y_true, y_pred))
|
|
71
|
+
|
|
72
|
+
print("\nAccuracy, Precision, Recall and F1:")
|
|
73
|
+
print(classification_report(y_true, y_pred))
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
import tensorflow as tf
|
|
2
|
+
import numpy as np
|
|
3
|
+
import matplotlib.pyplot as plt
|
|
4
|
+
|
|
5
|
+
from tensorflow.keras.applications import VGG16
|
|
6
|
+
from tensorflow.keras.applications.vgg16 import preprocess_input
|
|
7
|
+
from tensorflow.keras.preprocessing.text import Tokenizer
|
|
8
|
+
from tensorflow.keras.preprocessing.sequence import pad_sequences
|
|
9
|
+
from tensorflow.keras.layers import Input, Dense, LSTM, Embedding
|
|
10
|
+
from tensorflow.keras.models import Model
|
|
11
|
+
|
|
12
|
+
# Download Imagenette
|
|
13
|
+
url = "https://s3.amazonaws.com/fast-ai-imageclas/imagenette2-160.tgz"
|
|
14
|
+
|
|
15
|
+
archive = tf.keras.utils.get_file(
|
|
16
|
+
"imagenette2-160.tgz",
|
|
17
|
+
url,
|
|
18
|
+
untar=True
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
# Load 10 images
|
|
22
|
+
dataset = tf.keras.utils.image_dataset_from_directory(
|
|
23
|
+
archive + "/imagenette2-160/val",
|
|
24
|
+
image_size=(224, 224),
|
|
25
|
+
batch_size=10,
|
|
26
|
+
shuffle=True
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
images, labels = next(iter(dataset))
|
|
30
|
+
|
|
31
|
+
# ImageNet class names
|
|
32
|
+
names = {
|
|
33
|
+
"n01440764": "fish",
|
|
34
|
+
"n02102040": "dog",
|
|
35
|
+
"n02979186": "cassette player",
|
|
36
|
+
"n03000684": "chainsaw",
|
|
37
|
+
"n03028079": "church",
|
|
38
|
+
"n03394916": "French horn",
|
|
39
|
+
"n03417042": "garbage truck",
|
|
40
|
+
"n03425413": "gas pump",
|
|
41
|
+
"n03445777": "golf ball",
|
|
42
|
+
"n03888257": "parachute"
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
# Create captions
|
|
46
|
+
captions = []
|
|
47
|
+
|
|
48
|
+
for label in labels:
|
|
49
|
+
name = names[dataset.class_names[label]]
|
|
50
|
+
captions.append("start a photo of a " + name + " end")
|
|
51
|
+
|
|
52
|
+
print("Example:", captions[0])
|
|
53
|
+
|
|
54
|
+
# Tokenize
|
|
55
|
+
tokenizer = Tokenizer()
|
|
56
|
+
tokenizer.fit_on_texts(captions)
|
|
57
|
+
|
|
58
|
+
sequences = tokenizer.texts_to_sequences(captions)
|
|
59
|
+
|
|
60
|
+
vocab_size = len(tokenizer.word_index) + 1
|
|
61
|
+
max_length = max(len(x) for x in sequences)
|
|
62
|
+
|
|
63
|
+
print("Vocabulary size:", vocab_size)
|
|
64
|
+
|
|
65
|
+
# CNN feature extraction
|
|
66
|
+
cnn = VGG16(
|
|
67
|
+
weights="imagenet",
|
|
68
|
+
include_top=False,
|
|
69
|
+
pooling="avg"
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
cnn.trainable = False
|
|
73
|
+
|
|
74
|
+
features = cnn.predict(
|
|
75
|
+
preprocess_input(images),
|
|
76
|
+
verbose=0
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
print("Feature shape:", features.shape)
|
|
80
|
+
|
|
81
|
+
# Create training examples
|
|
82
|
+
X_image = []
|
|
83
|
+
X_text = []
|
|
84
|
+
Y = []
|
|
85
|
+
|
|
86
|
+
for i in range(len(sequences)):
|
|
87
|
+
|
|
88
|
+
seq = sequences[i]
|
|
89
|
+
|
|
90
|
+
for j in range(1, len(seq)):
|
|
91
|
+
|
|
92
|
+
X_image.append(features[i])
|
|
93
|
+
|
|
94
|
+
X_text.append(
|
|
95
|
+
seq[:j]
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
Y.append(
|
|
99
|
+
seq[j]
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
X_image = np.array(X_image)
|
|
103
|
+
|
|
104
|
+
X_text = pad_sequences(
|
|
105
|
+
X_text,
|
|
106
|
+
maxlen=max_length,
|
|
107
|
+
padding="pre"
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
Y = np.array(Y)
|
|
111
|
+
|
|
112
|
+
# LSTM decoder
|
|
113
|
+
image_input = Input(shape=(512,))
|
|
114
|
+
text_input = Input(shape=(max_length,))
|
|
115
|
+
|
|
116
|
+
image = Dense(128, activation="tanh")(image_input)
|
|
117
|
+
|
|
118
|
+
text = Embedding(
|
|
119
|
+
vocab_size,
|
|
120
|
+
128
|
|
121
|
+
)(text_input)
|
|
122
|
+
|
|
123
|
+
lstm = LSTM(128)(text, initial_state=[image, image])
|
|
124
|
+
|
|
125
|
+
output = Dense(
|
|
126
|
+
vocab_size,
|
|
127
|
+
activation="softmax"
|
|
128
|
+
)(lstm)
|
|
129
|
+
|
|
130
|
+
model = Model(
|
|
131
|
+
[image_input, text_input],
|
|
132
|
+
output
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
model.compile(
|
|
136
|
+
optimizer="adam",
|
|
137
|
+
loss="sparse_categorical_crossentropy",
|
|
138
|
+
metrics=["accuracy"]
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
# Train
|
|
142
|
+
model.fit(
|
|
143
|
+
[X_image, X_text],
|
|
144
|
+
Y,
|
|
145
|
+
epochs=100,
|
|
146
|
+
verbose=0
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
print("Training complete")
|
|
150
|
+
|
|
151
|
+
# Generate caption
|
|
152
|
+
feature = features[0:1]
|
|
153
|
+
|
|
154
|
+
words = ["start"]
|
|
155
|
+
|
|
156
|
+
for i in range(max_length):
|
|
157
|
+
|
|
158
|
+
sequence = tokenizer.texts_to_sequences(
|
|
159
|
+
[" ".join(words)]
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
sequence = pad_sequences(
|
|
163
|
+
sequence,
|
|
164
|
+
maxlen=max_length,
|
|
165
|
+
padding="pre"
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
prediction = model.predict(
|
|
169
|
+
[feature, sequence],
|
|
170
|
+
verbose=0
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
word_id = np.argmax(prediction[0])
|
|
174
|
+
|
|
175
|
+
word = tokenizer.index_word.get(
|
|
176
|
+
word_id,
|
|
177
|
+
""
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
if word == "end" or word == "":
|
|
181
|
+
break
|
|
182
|
+
|
|
183
|
+
words.append(word)
|
|
184
|
+
|
|
185
|
+
caption = " ".join(words[1:])
|
|
186
|
+
|
|
187
|
+
print("Generated Caption:", caption)
|
|
188
|
+
|
|
189
|
+
# Display image
|
|
190
|
+
plt.imshow(
|
|
191
|
+
images[0].numpy().astype("uint8")
|
|
192
|
+
)
|
|
193
|
+
|
|
194
|
+
plt.title(
|
|
195
|
+
caption
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
plt.axis("off")
|
|
199
|
+
plt.show()
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import tensorflow as tf
|
|
2
|
+
import numpy as np
|
|
3
|
+
import matplotlib.pyplot as plt
|
|
4
|
+
from tensorflow.keras import layers
|
|
5
|
+
|
|
6
|
+
# Text representation
|
|
7
|
+
texts = ["dog", "tiger", "bird", "truck", "church"]
|
|
8
|
+
text = np.eye(5)
|
|
9
|
+
|
|
10
|
+
# Generator: noise + text -> image
|
|
11
|
+
generator = tf.keras.Sequential([
|
|
12
|
+
layers.Input(shape=(105,)),
|
|
13
|
+
layers.Dense(128, activation="relu"),
|
|
14
|
+
layers.Dense(64 * 64 * 3, activation="sigmoid"),
|
|
15
|
+
layers.Reshape((64, 64, 3))
|
|
16
|
+
])
|
|
17
|
+
|
|
18
|
+
# Discriminator: image + text -> real/fake
|
|
19
|
+
image = layers.Input(shape=(64, 64, 3))
|
|
20
|
+
condition = layers.Input(shape=(5,))
|
|
21
|
+
|
|
22
|
+
x = layers.Flatten()(image)
|
|
23
|
+
x = layers.Concatenate()([x, condition])
|
|
24
|
+
x = layers.Dense(64, activation="relu")(x)
|
|
25
|
+
output = layers.Dense(1, activation="sigmoid")(x)
|
|
26
|
+
|
|
27
|
+
discriminator = tf.keras.Model([image, condition], output)
|
|
28
|
+
|
|
29
|
+
discriminator.compile(
|
|
30
|
+
optimizer="adam",
|
|
31
|
+
loss="binary_crossentropy"
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
# Random noise + text
|
|
35
|
+
noise = np.random.normal(0, 1, (5, 100))
|
|
36
|
+
fake_images = generator.predict(
|
|
37
|
+
np.concatenate([noise, text], axis=1),
|
|
38
|
+
verbose=0
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
# Train discriminator
|
|
42
|
+
real_images = np.random.random((5, 64, 64, 3))
|
|
43
|
+
|
|
44
|
+
discriminator.train_on_batch(
|
|
45
|
+
[real_images, text],
|
|
46
|
+
np.ones((5, 1))
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
discriminator.train_on_batch(
|
|
50
|
+
[fake_images, text],
|
|
51
|
+
np.zeros((5, 1))
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
print("Generated image shape:", fake_images.shape)
|
|
55
|
+
print("Discriminator output:",
|
|
56
|
+
discriminator.predict([fake_images, text], verbose=0))
|
|
57
|
+
|
|
58
|
+
# Unseen text
|
|
59
|
+
new_text = np.array([[1, 0, 0, 0, 0]])
|
|
60
|
+
new_noise = np.random.normal(0, 1, (1, 100))
|
|
61
|
+
|
|
62
|
+
new_image = generator.predict(
|
|
63
|
+
np.concatenate([new_noise, new_text], axis=1),
|
|
64
|
+
verbose=0
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
plt.imshow(new_image[0])
|
|
68
|
+
plt.title("Generated image for unseen text")
|
|
69
|
+
plt.axis("off")
|
|
70
|
+
plt.show()
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import tensorflow as tf
|
|
2
|
+
import numpy as np
|
|
3
|
+
import matplotlib.pyplot as plt
|
|
4
|
+
|
|
5
|
+
from tensorflow.keras.applications import VGG16
|
|
6
|
+
from tensorflow.keras.applications.vgg16 import preprocess_input
|
|
7
|
+
from tensorflow.keras.applications.vgg16 import decode_predictions
|
|
8
|
+
|
|
9
|
+
print("TensorFlow version:", tf.__version__)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
#============================
|
|
13
|
+
|
|
14
|
+
url = "https://s3.amazonaws.com/fast-ai-imageclas/imagenette2-160.tgz"
|
|
15
|
+
|
|
16
|
+
archive = tf.keras.utils.get_file(
|
|
17
|
+
"imagenette2-160.tgz",
|
|
18
|
+
url,
|
|
19
|
+
untar=True
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
print(archive)
|
|
23
|
+
|
|
24
|
+
#=================================
|
|
25
|
+
|
|
26
|
+
dataset = tf.keras.utils.image_dataset_from_directory(
|
|
27
|
+
archive + "/imagenette2-160/val",
|
|
28
|
+
image_size=(224, 224),
|
|
29
|
+
batch_size=5,
|
|
30
|
+
shuffle=True
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
print(dataset.class_names)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
#================================
|
|
37
|
+
|
|
38
|
+
images, labels = next(iter(dataset))
|
|
39
|
+
|
|
40
|
+
print("Original shape:", images.shape)
|
|
41
|
+
print("Preprocessed shape: (5, 224, 224, 3)")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
#=====================================
|
|
45
|
+
|
|
46
|
+
plt.figure(figsize=(12, 5))
|
|
47
|
+
|
|
48
|
+
for i in range(5):
|
|
49
|
+
plt.subplot(1, 5, i + 1)
|
|
50
|
+
plt.imshow(images[i].numpy().astype("uint8"))
|
|
51
|
+
plt.title(dataset.class_names[labels[i]])
|
|
52
|
+
plt.axis("off")
|
|
53
|
+
|
|
54
|
+
plt.show()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
#======================================
|
|
58
|
+
|
|
59
|
+
model = VGG16(weights="imagenet")
|
|
60
|
+
|
|
61
|
+
# Preprocess images
|
|
62
|
+
x = preprocess_input(images)
|
|
63
|
+
|
|
64
|
+
# Show preprocessed representation
|
|
65
|
+
print("Pixel value before preprocessing:", images[0][0][0])
|
|
66
|
+
print("Pixel value after preprocessing :", x[0][0][0])
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
#======================================
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# Predict
|
|
74
|
+
predictions = model.predict(x)
|
|
75
|
+
|
|
76
|
+
results = decode_predictions(predictions, top=5)
|
|
77
|
+
|
|
78
|
+
# Display predictions and compare with original labels
|
|
79
|
+
for i in range(5):
|
|
80
|
+
print("\nImage", i + 1)
|
|
81
|
+
print("Original label:", dataset.class_names[labels[i]])
|
|
82
|
+
print("Top-5 predictions:")
|
|
83
|
+
|
|
84
|
+
for _, label, probability in results[i]:
|
|
85
|
+
print(label, ":", round(probability * 100, 2), "%")
|
|
86
|
+
|
|
87
|
+
print("Predicted label:", results[i][0][1])
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tensorflow-uit2721dlca
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Deep Learning assignment
|
|
5
|
+
Requires-Python: <3.11,>=3.9
|
|
6
|
+
Description-Content-Type: text/markdown
|
|
7
|
+
Requires-Dist: tensorflow<2.10.0,>=2.9.0
|
|
8
|
+
Requires-Dist: numpy<1.22.0,>=1.21.6
|
|
9
|
+
Requires-Dist: matplotlib<3.6.0,>=3.5.2
|
|
10
|
+
Requires-Dist: scikit-learn<2.0.0,>=1.0.0
|
|
11
|
+
|
|
12
|
+
# tensorflow-uit2721dlca
|
|
13
|
+
|
|
14
|
+
Deep Learning assignment package — UIT2721DLCA.
|
|
15
|
+
|
|
16
|
+
## Requirements
|
|
17
|
+
|
|
18
|
+
- Python 3.9 or 3.10 (TensorFlow 2.9 does not support Python 3.11+)
|
|
19
|
+
|
|
20
|
+
## Installation
|
|
21
|
+
|
|
22
|
+
`ash
|
|
23
|
+
pip install tensorflow-uit2721dlca
|
|
24
|
+
`
|
|
25
|
+
|
|
26
|
+
## Usage
|
|
27
|
+
|
|
28
|
+
Each module is a self-contained assignment exercise. Importing a module will
|
|
29
|
+
execute its code (download data, train model, display plots).
|
|
30
|
+
|
|
31
|
+
`python
|
|
32
|
+
# Run exercise 1 — VGG16 image classification on Imagenette
|
|
33
|
+
import tensorflow_uit2721dlca.ex1
|
|
34
|
+
|
|
35
|
+
# Run exercise 2 — VGG16 feature maps and filters
|
|
36
|
+
import tensorflow_uit2721dlca.ex2
|
|
37
|
+
|
|
38
|
+
# Run exercise 3 — Transfer learning / fine-tuning
|
|
39
|
+
import tensorflow_uit2721dlca.ex3
|
|
40
|
+
|
|
41
|
+
# Run exercise 4 — Image captioning with CNN + LSTM
|
|
42
|
+
import tensorflow_uit2721dlca.ex4
|
|
43
|
+
|
|
44
|
+
# Run exercise 5 — Conditional GAN (text-to-image)
|
|
45
|
+
import tensorflow_uit2721dlca.ex5
|
|
46
|
+
|
|
47
|
+
# Run the prerequisites / environment check
|
|
48
|
+
import tensorflow_uit2721dlca.prerequisites
|
|
49
|
+
`
|
|
50
|
+
|
|
51
|
+
> **Note:** The exercises download the Imagenette dataset (~100 MB) on first
|
|
52
|
+
> run via f.keras.utils.get_file. The file is cached in ~/.keras/datasets/
|
|
53
|
+
> and is not re-downloaded on subsequent runs.
|
|
54
|
+
|
|
55
|
+
## Package name vs import name
|
|
56
|
+
|
|
57
|
+
| Purpose | Name |
|
|
58
|
+
|---|---|
|
|
59
|
+
| pip install | ensorflow-uit2721dlca |
|
|
60
|
+
| Python import | ensorflow_uit2721dlca |
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
src/tensorflow_uit2721dlca/__init__.py
|
|
4
|
+
src/tensorflow_uit2721dlca/ex1.py
|
|
5
|
+
src/tensorflow_uit2721dlca/ex2.py
|
|
6
|
+
src/tensorflow_uit2721dlca/ex3.py
|
|
7
|
+
src/tensorflow_uit2721dlca/ex4.py
|
|
8
|
+
src/tensorflow_uit2721dlca/ex5.py
|
|
9
|
+
src/tensorflow_uit2721dlca/prerequisites.py
|
|
10
|
+
src/tensorflow_uit2721dlca.egg-info/PKG-INFO
|
|
11
|
+
src/tensorflow_uit2721dlca.egg-info/SOURCES.txt
|
|
12
|
+
src/tensorflow_uit2721dlca.egg-info/dependency_links.txt
|
|
13
|
+
src/tensorflow_uit2721dlca.egg-info/requires.txt
|
|
14
|
+
src/tensorflow_uit2721dlca.egg-info/top_level.txt
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
tensorflow_uit2721dlca
|