Compare commits
349
Commits
93fb18fde2
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
70492acafe | ||
|
|
06913064c3 | ||
|
|
7190b6c438 | ||
|
|
7ce7a368c7 | ||
|
|
a2b818bbcc | ||
|
|
08062ad87d | ||
|
|
7176dfdf90 | ||
|
|
6399b88ade | ||
|
|
c5bcd09bba | ||
|
|
7d8af1c582 | ||
|
|
a379270add | ||
|
|
0b54ccaadb | ||
|
|
ac8ca830e1 | ||
|
|
49d2c48d31 | ||
|
|
280dbb579e | ||
|
|
ab7e346939 | ||
|
|
ad82ad4591 | ||
|
|
400706e730 | ||
|
|
096ffd56e9 | ||
|
|
790204af6a | ||
|
|
14a4125688 | ||
|
|
62519c371e | ||
|
|
9c8089d091 | ||
|
|
77a3219f39 | ||
|
|
66370a33c2 | ||
|
|
5870c3b0a3 | ||
|
|
e9340f17c4 | ||
|
|
282fee0806 | ||
|
|
33d30941a6 | ||
|
|
cb8d17cbc3 | ||
|
|
df2c2b7e5f | ||
|
|
6756ae4871 | ||
|
|
fa30aae3a8 | ||
|
|
15f3944d27 | ||
|
|
e6b73fe4fc | ||
|
|
1f72a2db8d | ||
|
|
a610bf4d89 | ||
|
|
9af3d9bc54 | ||
|
|
bbbeb403a8 | ||
|
|
9be7ec9e2e | ||
|
|
a6c717a401 | ||
|
|
4db59d2801 | ||
|
|
3a0c1e9067 | ||
|
|
5a8350d0ad | ||
|
|
62b6224de7 | ||
|
|
68bfff87a7 | ||
|
|
ac21c7a111 | ||
|
|
a18b4d304c | ||
|
|
a2f892e35c | ||
|
|
4bcfbbed0a | ||
|
|
31be0d7aac | ||
|
|
d4c9738db4 | ||
|
|
813daa0c90 | ||
|
|
29bf41c7da | ||
|
|
d9bd5be46d | ||
|
|
c2a2c34967 | ||
|
|
9e66981d3e | ||
|
|
6840ac72f1 | ||
|
|
ee304f1a75 | ||
|
|
4827a3f7dd | ||
|
|
d92051f5c4 | ||
|
|
10ce0d9124 | ||
|
|
88a20c4b0b | ||
|
|
090a1cd8bb | ||
|
|
ac6a305da8 | ||
|
|
cbedc395bc | ||
|
|
0c762b55b2 | ||
|
|
99b5820ae6 | ||
|
|
8214bd2669 | ||
|
|
aff0ae8fc6 | ||
|
|
16ad2b80ee | ||
|
|
88698dd735 | ||
|
|
f28bff7e47 | ||
|
|
9ee27b65a8 | ||
|
|
de4f984110 | ||
|
|
3b677e5e97 | ||
|
|
226fb1a206 | ||
|
|
7a1e17c29f | ||
|
|
f52720685e | ||
|
|
a3da7b048d | ||
|
|
f61eb59afb | ||
|
|
a430366bbe | ||
|
|
06d9753c46 | ||
|
|
4e6499cee3 | ||
|
|
7da3482366 | ||
|
|
7b3374d451 | ||
|
|
b17c1745ab | ||
|
|
db4ce6b425 | ||
|
|
aa6a097e4f | ||
|
|
150a213ae2 | ||
|
|
d860b71b1a | ||
|
|
1a3ee96932 | ||
|
|
8e366f26bc | ||
|
|
8a602264e6 | ||
|
|
99d5c88c8d | ||
|
|
c9da7378a1 | ||
|
|
b2773ecf09 | ||
|
|
084a13426a | ||
|
|
499996329f | ||
|
|
ce08ee4ccd | ||
|
|
cbbb044177 | ||
|
|
c630498168 | ||
|
|
f9d23c5bd4 | ||
|
|
3fa79faa9e | ||
|
|
556e26f21d | ||
|
|
01ee4664ff | ||
|
|
169e0531d9 | ||
|
|
3c46275015 | ||
|
|
0b86acd1a9 | ||
|
|
cddfced177 | ||
|
|
53d179663b | ||
|
|
6be837025b | ||
|
|
2854f74db4 | ||
|
|
462f7c79b4 | ||
|
|
0c9537975c | ||
|
|
10bfac73bd | ||
|
|
bdf3e73ba9 | ||
|
|
42400d6f32 | ||
|
|
0676011f98 | ||
|
|
e189b7c1f7 | ||
|
|
dc535167fa | ||
|
|
b722ddcb83 | ||
|
|
b755a81ef9 | ||
|
|
ae18041d6e | ||
|
|
d8e3d077e3 | ||
|
|
b54e2ed541 | ||
|
|
30f37b6ca5 | ||
|
|
860150905d | ||
|
|
66c3f38dba | ||
|
|
1a6cf69346 | ||
|
|
6f06485776 | ||
|
|
c7fd78dcf8 | ||
|
|
7a8eb93c48 | ||
|
|
4ba68d056b | ||
|
|
ba7a7299c2 | ||
|
|
6dba06704c | ||
|
|
20831c9bc3 | ||
|
|
e2338ad710 | ||
|
|
d1ee43e135 | ||
|
|
909f8f220c | ||
|
|
7fa88166ee | ||
|
|
321defb7ff | ||
|
|
00f1c07cfa | ||
|
|
312f525a46 | ||
|
|
78363988f7 | ||
|
|
b109ddb2b8 | ||
|
|
5b446a9f2f | ||
|
|
b462056e24 | ||
|
|
bdc0535ea4 | ||
|
|
5a1de4f2b2 | ||
|
|
4e499da168 | ||
|
|
8bfdec067b | ||
|
|
9d45f8a32b | ||
|
|
082ccf8382 | ||
|
|
8f10c16f3b | ||
|
|
5a674603ca | ||
|
|
ba0fc050fe | ||
|
|
7fc4d20262 | ||
|
|
d6036a95a3 | ||
|
|
14ccaace95 | ||
|
|
6d327bee22 | ||
|
|
270dc8e4d1 | ||
|
|
e49a0c37a6 | ||
|
|
a0726857a0 | ||
|
|
bf44947024 | ||
|
|
4f22b9755c | ||
|
|
ae3ac714d4 | ||
|
|
ad069269ac | ||
|
|
a54e7bbbb4 | ||
|
|
1f62054f36 | ||
|
|
59048a6862 | ||
|
|
a4764b8fc5 | ||
|
|
bf4cc9c483 | ||
|
|
f35cdefc21 | ||
|
|
493da70ba4 | ||
|
|
b415137e25 | ||
|
|
cead3d24c8 | ||
|
|
0bfbb1920c | ||
|
|
164102557e | ||
|
|
cf923a228d | ||
|
|
9f4a09a18c | ||
|
|
849b59e575 | ||
|
|
4a7e33b527 | ||
|
|
e5fad6d765 | ||
|
|
be6833ac6a | ||
|
|
6541a47668 | ||
|
|
96ccdb5c90 | ||
|
|
6e9c0a2626 | ||
|
|
396bf29f32 | ||
|
|
05774364fc | ||
|
|
74a9538feb | ||
|
|
9423c3378a | ||
|
|
e6fdbfe0b0 | ||
|
|
6c354b23bd | ||
|
|
3ad6c161b1 | ||
|
|
a1ff6a6f33 | ||
|
|
cf4f67d7a5 | ||
|
|
7c968db8ac | ||
|
|
e656cd4e12 | ||
|
|
8b1893b30a | ||
|
|
96d868a40c | ||
|
|
e698d058ff | ||
|
|
8aafbdae6c | ||
|
|
ec4162e34d | ||
|
|
99b7067cfe | ||
|
|
e62435cf22 | ||
|
|
3560a61277 | ||
|
|
7f0cbea937 | ||
|
|
31156eea54 | ||
|
|
ef11c33675 | ||
|
|
89aecfdef7 | ||
|
|
68e04c7653 | ||
|
|
a12f8c46e8 | ||
|
|
ff88dc1719 | ||
|
|
6502ea043a | ||
|
|
9cd4f70194 | ||
|
|
2cf55efcb3 | ||
|
|
d55adf87ac | ||
|
|
a621525d9e | ||
|
|
ba1ba027f0 | ||
|
|
24e1c44f5e | ||
|
|
94801efa6a | ||
|
|
4f7e8150e9 | ||
|
|
e9b5250e95 | ||
|
|
c1f38e3d77 | ||
|
|
9a107c7bb7 | ||
|
|
74d7307219 | ||
|
|
bdd0dfff10 | ||
|
|
e913bb28cc | ||
|
|
f9b845abd9 | ||
|
|
d46ad01e1d | ||
|
|
8d887b40b3 | ||
|
|
2ee5ab2b9c | ||
|
|
962d028c07 | ||
|
|
eaca23f27a | ||
|
|
0bd4008e72 | ||
|
|
123cab2500 | ||
|
|
d09cb9dd20 | ||
|
|
e1075b3551 | ||
|
|
d537cdf11d | ||
|
|
b8061466ce | ||
|
|
119282fee9 | ||
|
|
2e060f0776 | ||
|
|
9d0648f3c1 | ||
|
|
b093154193 | ||
|
|
f3aab68d5f | ||
|
|
0ebf7e6ce1 | ||
|
|
7229773371 | ||
|
|
ef985f6687 | ||
|
|
599769a8f6 | ||
|
|
70dca8f598 | ||
|
|
28d86e8762 | ||
|
|
fb278b641c | ||
|
|
3eb5f5e826 | ||
|
|
4a4ee95e31 | ||
|
|
4f8f32cf93 | ||
|
|
4e7f6cdee6 | ||
|
|
6a36215d82 | ||
|
|
8620d43e74 | ||
|
|
d26eeaddd1 | ||
|
|
b9d883484f | ||
|
|
833724912e | ||
|
|
717bb352e9 | ||
|
|
cb8e907ecb | ||
|
|
084273f5bc | ||
|
|
b98234267f | ||
|
|
11aed4497c | ||
|
|
93af7a69be | ||
|
|
385eb79f0e | ||
|
|
dcd01849ec | ||
|
|
cf03275da0 | ||
|
|
cecdf546b8 | ||
|
|
bd9c8136d1 | ||
|
|
675835a56b | ||
|
|
68a542be04 | ||
|
|
2ab4cf3b71 | ||
|
|
0fbdc3865c | ||
|
|
d516578d4b | ||
|
|
1579b15b4e | ||
|
|
9fbed5a9c4 | ||
|
|
6f0513ea2c | ||
|
|
0bbf53d495 | ||
|
|
c6f4c81ddd | ||
|
|
1f4f9949fe | ||
|
|
8c5b58ebf3 | ||
|
|
8db61b2b8a | ||
|
|
a500e8a687 | ||
|
|
7d723285c8 | ||
|
|
75e61bf05c | ||
|
|
291407dfac | ||
|
|
8724bda2f1 | ||
|
|
4a6d0dbe09 | ||
|
|
fd9140093d | ||
|
|
72c60170a7 | ||
|
|
e3e836c9fa | ||
|
|
9a625cb6be | ||
|
|
38fc595725 | ||
|
|
e04ce33891 | ||
|
|
bc653a0be4 | ||
|
|
d9d2dbab74 | ||
|
|
f10f9660a3 | ||
|
|
649dedeb29 | ||
|
|
5391927c58 | ||
|
|
7abd4b7403 | ||
|
|
7d86b75833 | ||
|
|
3d6207bf57 | ||
|
|
df2610dc22 | ||
|
|
1975b13151 | ||
|
|
12d043b220 | ||
|
|
2ff5124239 | ||
|
|
f983145dfb | ||
|
|
0a3a07da12 | ||
|
|
c3aba7831b | ||
|
|
75b57bd28f | ||
|
|
bb95e78132 | ||
|
|
eb4185c129 | ||
|
|
e09188453e | ||
|
|
a49e55c159 | ||
|
|
2a85997aa7 | ||
|
|
b430a15ece | ||
|
|
abd83aa542 | ||
|
|
a2164b9ccf | ||
|
|
0e404015f1 | ||
|
|
e0eb1dd53c | ||
|
|
28c9d12a5b | ||
|
|
167f9ecb3c | ||
|
|
dc976540bf | ||
|
|
133019ad22 | ||
|
|
dc9f611e17 | ||
|
|
657e9c4b1d | ||
|
|
080c7e7a5c | ||
|
|
8965eb7192 | ||
|
|
6fec85adfe | ||
|
|
a7f2f67d05 | ||
|
|
7f906068d7 | ||
|
|
5021cfec51 | ||
|
|
68c2bb56d5 | ||
|
|
ccf2abda63 | ||
|
|
5651c9d200 | ||
|
|
7b3782a1ce | ||
|
|
eceb6fba9e | ||
|
|
50756450e4 | ||
|
|
54eca39b6d | ||
|
|
b9f333c7d0 | ||
|
|
865a8ff2b7 | ||
|
|
08c5c14da6 | ||
|
|
f139206758 | ||
|
|
ec48879327 | ||
|
|
1065a35cf0 |
@@ -1,27 +0,0 @@
|
||||
name: Code Quality Pipeline
|
||||
run-name: ${{ gitea.actor }} is running the Code Quality Pipeline
|
||||
runs-on: ubuntu-latest
|
||||
on: push
|
||||
image: python:3.12
|
||||
jobs:
|
||||
test:
|
||||
name: Test
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout Code
|
||||
uses: actions/checkout@v3
|
||||
- name: Setup Environment
|
||||
uses: https://github.com/actions/setup-python@v3
|
||||
with:
|
||||
python-verison: "3.12"
|
||||
architecture: "x64"
|
||||
- name: Install Packages
|
||||
run: |
|
||||
pip install poetry
|
||||
poetry install
|
||||
- name: PEP8 Check
|
||||
run: |
|
||||
poetry run flake8 ./src --benchmark
|
||||
- name: Type Check
|
||||
run: |
|
||||
poetry run mypy ./src --disable-error-code=import-untyped
|
||||
@@ -1,10 +1,8 @@
|
||||
name: CI Pipeline
|
||||
run-name: ${{ gitea.actor }} is running the CI Pipeline
|
||||
runs-on: ubuntu-latest
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
release:
|
||||
types: [published]
|
||||
jobs:
|
||||
test:
|
||||
name: Test
|
||||
@@ -18,35 +16,60 @@ jobs:
|
||||
python-verison: "3.12"
|
||||
architecture: "x64"
|
||||
- name: Install Packages
|
||||
env:
|
||||
PIP_INDEX_URL: http://192.168.1.2:5001/index/
|
||||
PIP_TRUSTED_HOST: 192.168.1.2
|
||||
run: |
|
||||
pip install poetry
|
||||
poetry install
|
||||
- name: PEP8 Check
|
||||
run: |
|
||||
poetry run flake8 ./src --benchmark
|
||||
poetry run flake8 . --benchmark
|
||||
- name: Type Check
|
||||
run: |
|
||||
poetry run mypy ./src --disable-error-code=import-untyped
|
||||
poetry run mypy .
|
||||
publish:
|
||||
name: Build and Publish
|
||||
runs-on: ubuntu-latest
|
||||
needs: [test]
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
-
|
||||
dockerfile: ./Dockerfile.web_ui
|
||||
image: ${{ vars.docker_repo_url }}/${{ gitea.repository }}/web_ui
|
||||
-
|
||||
dockerfile: ./Dockerfile.model
|
||||
image: ${{ vars.docker_repo_url }}/${{ gitea.repository }}/model
|
||||
steps:
|
||||
- name: Checkout Code
|
||||
-
|
||||
name: Checkout Code
|
||||
uses: actions/checkout@v3
|
||||
- name: Set Environment Variables
|
||||
-
|
||||
name: Set Environment Variables
|
||||
run: |
|
||||
echo "sha_short=$(git rev-parse --short ${{ gitea.sha }} )" >> "$GITHUB_ENV"
|
||||
- name: Show Environment Variables
|
||||
-
|
||||
name: Show Environment Variables
|
||||
run: |
|
||||
echo "DOCKER_REPO_URL: ${{ vars.docker_repo_url }}"
|
||||
echo "REPOSITORY: ${{ gitea.repository }}"
|
||||
echo "COMMIT_SHA: ${{ env.sha_short }}"
|
||||
- name: Build and Push Image
|
||||
uses: docker/build-push-action@v2
|
||||
-
|
||||
name: Extract metadata (tags, labels) for Docker
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: ${{ matrix.image }}
|
||||
-
|
||||
name: Build and Push Image
|
||||
uses: docker/build-push-action@v4
|
||||
with:
|
||||
context: .
|
||||
file: ${{ matrix.dockerfile }}
|
||||
push: true
|
||||
tags: |
|
||||
${{ vars.docker_repo_url }}/${{ gitea.repository }}:${{ env.sha_short }}
|
||||
${{ vars.docker_repo_url }}/${{ gitea.repository }}:latest
|
||||
${{ matrix.image }}:${{ env.sha_short }}
|
||||
${{ matrix.image }}:latest
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
name: Code Quality Pipeline
|
||||
on:
|
||||
pull_request:
|
||||
branches:
|
||||
- main
|
||||
jobs:
|
||||
job:
|
||||
name: Check Code
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout Code
|
||||
uses: actions/checkout@v4
|
||||
- name: Setup Environment
|
||||
uses: https://github.com/actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
architecture: "x64"
|
||||
- name: Setup poetry
|
||||
env:
|
||||
POETRY_VERSION: 2.1.1
|
||||
POETRY_HOME: /opt/poetry
|
||||
POETRY_NO_INTERACTION: 1
|
||||
POETRY_NO_CACHE: 1
|
||||
run: |
|
||||
curl -sSL https://install.python-poetry.org | python3 -
|
||||
export PATH=$POETRY_HOME/bin:$PATH
|
||||
poetry --version
|
||||
- name: Install Dependencies
|
||||
env:
|
||||
PIP_INDEX_URL: ${{ vars.PIP_INDEX_URL }}
|
||||
PIP_TRUSTED_HOST: ${{ vars.PIP_TRUSTED_HOST }}
|
||||
run: |
|
||||
/opt/poetry/bin/poetry install
|
||||
- name: PEP8 Check
|
||||
run: |
|
||||
/opt/poetry/bin/poetry run flake8 . --benchmark
|
||||
- name: Type Check
|
||||
run: |
|
||||
/opt/poetry/bin/poetry run mypy .
|
||||
- name: Pytest & Calculate Coverage
|
||||
env:
|
||||
MINIO_ENDPOINT: ${{ vars.MINIO_ENDPOINT }}
|
||||
MINIO_ACCESS_KEY: ${{ vars.MINIO_ACCESS_KEY }}
|
||||
MINIO_SECRET_KEY: ${{ secrets.MINIO_SECRET_KEY }}
|
||||
MONGO_ENDPOINT: ${{ vars.MONGO_ENDPOINT }}
|
||||
run: |
|
||||
/opt/poetry/bin/poetry run coverage run -m pytest .
|
||||
- name: Coverage Report
|
||||
run: |
|
||||
/opt/poetry/bin/poetry run coverage report -m
|
||||
@@ -0,0 +1,17 @@
|
||||
name: release-tag
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
version:
|
||||
description: 'The version number to deploy'
|
||||
required: true
|
||||
jobs:
|
||||
release-image:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
-
|
||||
run: echo "manual trigger executed"
|
||||
-
|
||||
run: echo "$MESSAGE"
|
||||
env:
|
||||
MESSAGE; ${{ github.event.inputs.version}}
|
||||
@@ -177,3 +177,6 @@ ipython_config.py
|
||||
|
||||
# Remove previous ipynb_checkpoints
|
||||
# git rm -r .ipynb_checkpoints/
|
||||
|
||||
# Logging data
|
||||
runs/*
|
||||
|
||||
+33
-15
@@ -1,44 +1,62 @@
|
||||
repos:
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: v4.5.0
|
||||
rev: v5.0.0
|
||||
hooks:
|
||||
- id: trailing-whitespace
|
||||
- id: end-of-file-fixer
|
||||
- id: check-yaml
|
||||
- id: check-added-large-files
|
||||
- id: debug-statements
|
||||
- id: double-quote-string-fixer
|
||||
- id: name-tests-test
|
||||
- id: requirements-txt-fixer
|
||||
- repo: https://github.com/asottile/setup-cfg-fmt
|
||||
rev: v2.5.0
|
||||
rev: v2.7.0
|
||||
hooks:
|
||||
- id: setup-cfg-fmt
|
||||
- repo: https://github.com/asottile/reorder-python-imports
|
||||
rev: v3.12.0
|
||||
- repo: https://github.com/pre-commit/mirrors-isort
|
||||
rev: v5.10.1
|
||||
hooks:
|
||||
- id: reorder-python-imports
|
||||
exclude: ^(pre_commit/resources/|testing/resources/python3_hooks_repo/)
|
||||
args: [--py39-plus, --add-import, 'from __future__ import annotations']
|
||||
- id: isort
|
||||
language_version: python3.10
|
||||
args: [ --tc ]
|
||||
- repo: https://github.com/asottile/add-trailing-comma
|
||||
rev: v3.1.0
|
||||
hooks:
|
||||
- id: add-trailing-comma
|
||||
- repo: https://github.com/asottile/pyupgrade
|
||||
rev: v3.15.1
|
||||
rev: v3.19.1
|
||||
hooks:
|
||||
- id: pyupgrade
|
||||
args: [--py39-plus]
|
||||
- repo: https://github.com/hhatto/autopep8
|
||||
rev: v2.0.4
|
||||
hooks:
|
||||
- id: autopep8
|
||||
- repo: https://github.com/PyCQA/flake8
|
||||
rev: 7.0.0
|
||||
hooks:
|
||||
- id: flake8
|
||||
entry: pflake8
|
||||
additional_dependencies:
|
||||
- "pyproject-flake8"
|
||||
- repo: https://github.com/pre-commit/mirrors-mypy
|
||||
rev: v1.8.0
|
||||
rev: v1.15.0
|
||||
hooks:
|
||||
- id: mypy
|
||||
additional_dependencies: [types-all]
|
||||
exclude: ^testing/resources/
|
||||
- repo: https://github.com/psf/black
|
||||
rev: 25.1.0
|
||||
hooks:
|
||||
- id: black
|
||||
language_version: python3.12
|
||||
args: [ --skip-string-normalization ]
|
||||
- repo: https://github.com/myint/docformatter
|
||||
rev: v1.5.0
|
||||
hooks:
|
||||
- id: docformatter
|
||||
name: docformatter
|
||||
description: 'Formats docstrings to follow PEP 257.'
|
||||
entry: docformatter
|
||||
language: python
|
||||
types: [ python ]
|
||||
- repo: https://github.com/jendrikseipp/vulture
|
||||
rev: 'v2.14'
|
||||
hooks:
|
||||
- id: vulture
|
||||
entry: vulture . --min-confidence 90 --exclude */.venv/*.py,*/tests/*.py
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
# build stage
|
||||
FROM python:3.12-slim-bookworm AS builder
|
||||
|
||||
# set environment variables
|
||||
ENV PYTHONUNBUFFERED=1 \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
PIP_NO_CACHE_DIR=off \
|
||||
PIP_DISABLE_PIP_VERSION_CHECK=ON \
|
||||
PIP_DEFAULT_TIMEOUT=100 \
|
||||
DEBIAN_FRONTEND=noninteractive \
|
||||
POETRY_HOME=/etc/poetry \
|
||||
POETRY_VERSION=1.7.1 \
|
||||
POETRY_VIRTUALENVS_IN_PROJECT=1 \
|
||||
POETRY_VIRTUALENVS_CREATE=1 \
|
||||
POETRY_NO_INTERACTION=1 \
|
||||
POETRY_CACHE_DIR=/tmp/poetry_cache \
|
||||
APP_HOME=/home/app
|
||||
|
||||
# update system
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
build-essential \
|
||||
curl \
|
||||
&& apt-get clean
|
||||
|
||||
# install poetry
|
||||
RUN curl -sSL https://install.python-poetry.org | python3 -
|
||||
ENV PATH="${POETRY_HOME}/bin:$PATH"
|
||||
|
||||
# install runtime dependencies
|
||||
WORKDIR ${APP_HOME}
|
||||
COPY ./poetry.lock ./pyproject.toml ./
|
||||
RUN --mount=type=cache,target=${POETRY_CACHE_DIR} poetry install \
|
||||
--with shared,model \
|
||||
--no-root
|
||||
|
||||
# final stage
|
||||
FROM python:3.12-slim-bookworm
|
||||
|
||||
# copy virtualenv made by poetry
|
||||
ENV APP_HOME=/home/app \
|
||||
VIRTUAL_ENV=/home/app/.venv
|
||||
COPY --from=builder ${VIRTUAL_ENV} ${VIRTUAL_ENV}
|
||||
ENV PATH="${VIRTUAL_ENV}/bin:${PATH}"
|
||||
|
||||
# create home directory and app user
|
||||
RUN mkdir -p $APP_HOME
|
||||
|
||||
# add code while changing ownership
|
||||
WORKDIR $APP_HOME
|
||||
COPY ./shared ./shared
|
||||
COPY ./model/src ./
|
||||
|
||||
ENTRYPOINT [ "python", "main.py" ]
|
||||
@@ -0,0 +1,59 @@
|
||||
# build stage
|
||||
FROM python:3.12-slim-bookworm AS builder
|
||||
|
||||
# set environment variables
|
||||
ENV PYTHONUNBUFFERED=1 \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
PIP_NO_CACHE_DIR=off \
|
||||
PIP_DISABLE_PIP_VERSION_CHECK=ON \
|
||||
PIP_DEFAULT_TIMEOUT=100 \
|
||||
DEBIAN_FRONTEND=noninteractive \
|
||||
POETRY_HOME=/etc/poetry \
|
||||
POETRY_VERSION=1.7.1 \
|
||||
POETRY_VIRTUALENVS_IN_PROJECT=1 \
|
||||
POETRY_VIRTUALENVS_CREATE=1 \
|
||||
POETRY_NO_INTERACTION=1 \
|
||||
POETRY_CACHE_DIR=/tmp/poetry_cache \
|
||||
APP_HOME=/home/app
|
||||
|
||||
# update system
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
build-essential \
|
||||
curl \
|
||||
&& apt-get clean
|
||||
|
||||
# install poetry
|
||||
RUN curl -sSL https://install.python-poetry.org | python3 -
|
||||
ENV PATH="${POETRY_HOME}/bin:$PATH"
|
||||
|
||||
# install runtime dependencies
|
||||
WORKDIR ${APP_HOME}
|
||||
COPY ./poetry.lock ./pyproject.toml ./
|
||||
RUN --mount=type=cache,target=${POETRY_CACHE_DIR} poetry install \
|
||||
--with shared,web_ui \
|
||||
--no-root
|
||||
|
||||
# final stage
|
||||
FROM python:3.12-slim-bookworm
|
||||
|
||||
# copy virtualenv made by poetry
|
||||
ENV APP_HOME=/home/app \
|
||||
VIRTUAL_ENV=/home/app/.venv
|
||||
COPY --from=builder ${VIRTUAL_ENV} ${VIRTUAL_ENV}
|
||||
ENV PATH="${VIRTUAL_ENV}/bin:${PATH}"
|
||||
|
||||
# create home directory and app user
|
||||
RUN mkdir -p /home/app && \
|
||||
addgroup --system app && \
|
||||
adduser --system --group app
|
||||
|
||||
# add code while changing ownership
|
||||
WORKDIR $APP_HOME
|
||||
COPY --chown=app:app ./shared ./shared
|
||||
COPY --chown=app:app ./web_ui/src ./src
|
||||
|
||||
# change to the app user
|
||||
USER app
|
||||
|
||||
ENTRYPOINT [ "gunicorn", "src.main:server", "-b", "0.0.0.0:8050" ]
|
||||
@@ -1,13 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from .classes import ModelOutputs
|
||||
from .classes import NoDocumentFoundException
|
||||
from .classes import VisualCommunication
|
||||
from .database import connect
|
||||
from .utils import get_visual_communication
|
||||
from .utils import list_names
|
||||
from .utils import total_annotated
|
||||
from .utils import total_documents
|
||||
from .utils import upsert_annotations
|
||||
from .utils import upsert_predictions
|
||||
from .utils import upsert_visual_communication
|
||||
@@ -1,165 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from base64 import b64decode
|
||||
from base64 import b64encode
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
|
||||
from PIL import Image
|
||||
from pydantic import BaseModel
|
||||
from pydantic import field_serializer
|
||||
from pydantic import field_validator
|
||||
|
||||
from src.model_experiential import (
|
||||
VisualSyntaxModelOutput,
|
||||
)
|
||||
from src.model_interpersonal import AngleModelOutput
|
||||
from src.model_interpersonal import ContactModelOutput
|
||||
from src.model_interpersonal import DistanceModelOutput
|
||||
from src.model_interpersonal import ModalityColorModelOutput
|
||||
from src.model_interpersonal import ModalityDepthModelOutput
|
||||
from src.model_interpersonal import ModalityLightingModelOutput
|
||||
from src.model_interpersonal import PointOfViewModelOutput
|
||||
from src.model_textual import FramingModelOutput
|
||||
from src.model_textual import InformationValueModelOutput
|
||||
from src.model_textual import SalienceModelOutput
|
||||
|
||||
|
||||
class NoDocumentFoundException(Exception):
|
||||
pass
|
||||
|
||||
|
||||
class ModelOutputs(BaseModel):
|
||||
visual_syntax: VisualSyntaxModelOutput
|
||||
contact: ContactModelOutput
|
||||
angle: AngleModelOutput
|
||||
point_of_view: PointOfViewModelOutput
|
||||
distance: DistanceModelOutput
|
||||
modality_lighting: ModalityLightingModelOutput
|
||||
modality_color: ModalityColorModelOutput
|
||||
modality_depth: ModalityDepthModelOutput
|
||||
information_value: InformationValueModelOutput
|
||||
framing: FramingModelOutput
|
||||
salience: SalienceModelOutput
|
||||
|
||||
@classmethod
|
||||
def list_fields(cls) -> list[str]:
|
||||
"""List options that are stored as attributes."""
|
||||
return list(cls.model_fields.keys())
|
||||
|
||||
@classmethod
|
||||
def from_random(cls) -> ModelOutputs:
|
||||
"""Instantiate with random numbers."""
|
||||
kwargs = {
|
||||
field: field_info.annotation.from_random() # type: ignore
|
||||
for field, field_info
|
||||
in cls.model_fields.items()
|
||||
}
|
||||
return cls(**kwargs)
|
||||
|
||||
@classmethod
|
||||
def from_annotations(
|
||||
cls,
|
||||
visual_syntax: str,
|
||||
contact: str,
|
||||
angle: str,
|
||||
point_of_view: str,
|
||||
distance: str,
|
||||
modality_lighting: str,
|
||||
modality_color: str,
|
||||
modality_depth: str,
|
||||
information_value: str,
|
||||
framing: str,
|
||||
salience: str,
|
||||
) -> ModelOutputs:
|
||||
"""Instantiate from annotation."""
|
||||
kwargs = {
|
||||
'visual_syntax': VisualSyntaxModelOutput
|
||||
.from_choice(visual_syntax),
|
||||
'contact': ContactModelOutput
|
||||
.from_choice(contact),
|
||||
'angle': AngleModelOutput
|
||||
.from_choice(angle),
|
||||
'point_of_view': PointOfViewModelOutput
|
||||
.from_choice(point_of_view),
|
||||
'distance': DistanceModelOutput
|
||||
.from_choice(distance),
|
||||
'modality_lighting': ModalityLightingModelOutput
|
||||
.from_choice(modality_lighting),
|
||||
'modality_color': ModalityColorModelOutput
|
||||
.from_choice(modality_color),
|
||||
'modality_depth': ModalityDepthModelOutput
|
||||
.from_choice(modality_depth),
|
||||
'information_value': InformationValueModelOutput
|
||||
.from_choice(information_value),
|
||||
'framing': FramingModelOutput
|
||||
.from_choice(framing),
|
||||
'salience': SalienceModelOutput
|
||||
.from_choice(salience),
|
||||
}
|
||||
return cls(**kwargs)
|
||||
|
||||
|
||||
class VisualCommunication(BaseModel):
|
||||
name: str
|
||||
image: Image.Image
|
||||
annotation: ModelOutputs | None = None
|
||||
prediction: ModelOutputs | None = None
|
||||
|
||||
class Config:
|
||||
arbitrary_types_allowed = True
|
||||
|
||||
@classmethod
|
||||
def classname(cls) -> str:
|
||||
"""Return classname."""
|
||||
return cls.__name__
|
||||
|
||||
@classmethod
|
||||
def from_file(cls, path: Path) -> VisualCommunication:
|
||||
"""Instantiate from file."""
|
||||
name = path.stem
|
||||
image = Image.open(path)
|
||||
image.load()
|
||||
return VisualCommunication(name=name, image=image)
|
||||
|
||||
@classmethod
|
||||
def decode_image(cls, content: str) -> Image.Image:
|
||||
"""Decode image."""
|
||||
_, content_data = content.split(',')
|
||||
return Image.open(BytesIO(b64decode(content_data)))
|
||||
|
||||
@field_serializer('image')
|
||||
def serialize_image(image: Image.Image) -> bytes: # type: ignore
|
||||
buffer = BytesIO()
|
||||
image.save(buffer, format='JPEG')
|
||||
return buffer.getvalue()
|
||||
|
||||
@field_validator('image', mode='before')
|
||||
@classmethod
|
||||
def convert_to_image(
|
||||
cls,
|
||||
image: Image.Image | BytesIO | bytes,
|
||||
) -> Image.Image:
|
||||
if isinstance(image, bytes):
|
||||
image = BytesIO(image)
|
||||
if isinstance(image, BytesIO):
|
||||
image = Image.open(image)
|
||||
return image
|
||||
|
||||
def __repr__(self) -> str:
|
||||
return f"{self.classname()}(name='{self.name}')"
|
||||
|
||||
def webencoded_image(self) -> str:
|
||||
"""Convert image to be displayed on webpage."""
|
||||
# convert images to bytes string
|
||||
buffer = BytesIO()
|
||||
self.image.save(buffer, format='png')
|
||||
img_enc = b64encode(buffer.getvalue()).decode('utf-8')
|
||||
return f"data:image/png;base64, {img_enc}"
|
||||
|
||||
def generate_random_prediction(self, force: bool = False) -> None:
|
||||
"""Generate random prediction values."""
|
||||
if not force and self.prediction is not None:
|
||||
logging.warning('set force=True to overwrite existing values.')
|
||||
self.prediction = ModelOutputs.from_random()
|
||||
@@ -1,27 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
|
||||
from dotenv import load_dotenv
|
||||
from pymongo import MongoClient
|
||||
|
||||
|
||||
def connect():
|
||||
"""Connect to MongoDB."""
|
||||
# load env vars
|
||||
load_dotenv()
|
||||
necessary_env_vars = [
|
||||
'MONGO_HOST',
|
||||
'MONGO_DB',
|
||||
'MONGO_COLLECTION',
|
||||
]
|
||||
for env_var in necessary_env_vars:
|
||||
assert env_var in os.environ, f"{env_var} not found"
|
||||
# connect to database
|
||||
client = MongoClient(os.getenv('MONGO_HOST'))
|
||||
db = client[os.getenv('MONGO_DB')]
|
||||
collection = db[os.getenv('MONGO_COLLECTION')]
|
||||
collection.create_index('name', unique=True)
|
||||
logging.info('connected to database')
|
||||
return collection, db, client
|
||||
@@ -1,158 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from pymongo.collection import Collection
|
||||
|
||||
from .classes import ModelOutputs
|
||||
from .classes import NoDocumentFoundException
|
||||
from .classes import VisualCommunication
|
||||
|
||||
|
||||
def count_documents(
|
||||
collection: Collection,
|
||||
has_annotation: bool = False,
|
||||
) -> int:
|
||||
"""
|
||||
Get the total number of documents
|
||||
in database that matches the filters.
|
||||
"""
|
||||
assert isinstance(collection, Collection)
|
||||
assert isinstance(has_annotation, bool)
|
||||
# build query
|
||||
query = {}
|
||||
if has_annotation:
|
||||
query['annotation'] = {'$ne': None}
|
||||
return collection.count_documents(filter=query)
|
||||
|
||||
|
||||
def list_names(
|
||||
collection: Collection,
|
||||
has_annotation: bool = True,
|
||||
) -> list[str]:
|
||||
"""List the names of entries that match the filters."""
|
||||
assert isinstance(collection, Collection)
|
||||
assert isinstance(has_annotation, bool)
|
||||
# build query
|
||||
query = {}
|
||||
if has_annotation:
|
||||
query['annotation'] = {'$ne': None}
|
||||
res_list = collection.find(
|
||||
filter=query,
|
||||
projection={
|
||||
'_id': False,
|
||||
'name': True,
|
||||
},
|
||||
)
|
||||
return list(res_list)
|
||||
|
||||
|
||||
def total_documents(
|
||||
collection: Collection,
|
||||
) -> int:
|
||||
"""Get total number of documents in database."""
|
||||
return collection.count_documents(filter={})
|
||||
|
||||
|
||||
def total_annotated(
|
||||
collection: Collection,
|
||||
) -> int:
|
||||
"""Get total number of annotated documents in database."""
|
||||
query = {
|
||||
'annotation': {
|
||||
'$ne': None,
|
||||
},
|
||||
}
|
||||
return collection.count_documents(filter=query)
|
||||
|
||||
|
||||
def get_visual_communication(
|
||||
collection: Collection,
|
||||
with_annotation: bool = False,
|
||||
) -> VisualCommunication:
|
||||
"""Get a random visual communication from the database."""
|
||||
query = {}
|
||||
if with_annotation:
|
||||
query['annotation'] = {'$ne': None}
|
||||
else:
|
||||
query['annotation'] = {'$eq': None}
|
||||
data = collection.aggregate([
|
||||
{
|
||||
'$match': query, # find using filters
|
||||
},
|
||||
{
|
||||
'$sample': {
|
||||
'size': 1, # get one random
|
||||
},
|
||||
},
|
||||
])
|
||||
data_list = list(data) # read data from cursor object
|
||||
if len(data_list) == 0:
|
||||
logging.error('failed getting visual communication')
|
||||
raise NoDocumentFoundException()
|
||||
logging.info('finished')
|
||||
return VisualCommunication.model_validate(data_list[0])
|
||||
|
||||
|
||||
def upsert_predictions(
|
||||
collection: Collection,
|
||||
vis_com_name: str,
|
||||
predictions: ModelOutputs,
|
||||
) -> None:
|
||||
"""Upsert prediction data in the database."""
|
||||
query = {
|
||||
'name': vis_com_name,
|
||||
}
|
||||
update = {
|
||||
'$set': {
|
||||
'prediction': predictions.model_dump(),
|
||||
},
|
||||
}
|
||||
res = collection.update_one(
|
||||
filter=query,
|
||||
update=update,
|
||||
upsert=True,
|
||||
)
|
||||
logging.debug('upserted document: %s', res)
|
||||
logging.info('finished')
|
||||
|
||||
|
||||
def upsert_annotations(
|
||||
collection: Collection,
|
||||
vis_com_name: str,
|
||||
annotations: ModelOutputs,
|
||||
) -> None:
|
||||
"""Upserts annotation data in the database."""
|
||||
query = {
|
||||
'name': vis_com_name,
|
||||
}
|
||||
update = {
|
||||
'$set': {
|
||||
'annotation': annotations.model_dump(),
|
||||
},
|
||||
}
|
||||
res = collection.update_one(
|
||||
filter=query,
|
||||
update=update,
|
||||
upsert=True,
|
||||
)
|
||||
logging.info('upserted document: %s', res)
|
||||
logging.info('finished')
|
||||
|
||||
|
||||
def upsert_visual_communication(
|
||||
collection: Collection,
|
||||
visual_communication_list: list[VisualCommunication],
|
||||
) -> bool:
|
||||
"""
|
||||
Upsert VisualCommunication object in the database.
|
||||
Returns bool stating success.
|
||||
"""
|
||||
response = collection.insert_many(
|
||||
[
|
||||
vis_com.model_dump()
|
||||
for vis_com
|
||||
in visual_communication_list
|
||||
],
|
||||
)
|
||||
return response.acknowledged
|
||||
@@ -1,11 +1,10 @@
|
||||
version: '3.7'
|
||||
services:
|
||||
app:
|
||||
image: visual_critical_discourse_analysis:dev
|
||||
container_name: visual_critical_discourse_analysis
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
dockerfile: ./web_ui/Dockerfile
|
||||
env_file:
|
||||
- local.env
|
||||
environment:
|
||||
|
||||
@@ -1,11 +1,10 @@
|
||||
version: '3.7'
|
||||
services:
|
||||
app:
|
||||
image: visual_critical_discourse_analysis:dev
|
||||
container_name: visual_critical_discourse_analysis
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
dockerfile: ./web_ui/Dockerfile
|
||||
env_file:
|
||||
- server.env
|
||||
ports:
|
||||
|
||||
@@ -6,10 +6,9 @@ from pathlib import Path
|
||||
|
||||
import pandas as pd
|
||||
import requests
|
||||
from classes import Instagram
|
||||
from retry import retry
|
||||
|
||||
from image_download.classes import Instagram
|
||||
|
||||
|
||||
def get_sources() -> pd.DataFrame:
|
||||
"""Get sources dateframe."""
|
||||
|
||||
@@ -3,10 +3,11 @@ from __future__ import annotations
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from database import connect
|
||||
from database import VisualCommunication
|
||||
from dotenv import load_dotenv
|
||||
|
||||
from shared.docstore import connect_mongodb
|
||||
from shared.docstore.classes import VisualCommunication
|
||||
|
||||
if __name__ == '__main__':
|
||||
# prepare env vars
|
||||
env_path = Path(__file__).parent.parent / 'local.env'
|
||||
@@ -14,7 +15,7 @@ if __name__ == '__main__':
|
||||
load_dotenv(env_path)
|
||||
os.environ['MONGO_HOST'] = 'localhost'
|
||||
# connect to database
|
||||
collection, db, client = connect()
|
||||
collection, db, client = connect_mongodb()
|
||||
print(client.server_info())
|
||||
# download images
|
||||
data = None
|
||||
|
Before Width: | Height: | Size: 26 KiB After Width: | Height: | Size: 26 KiB |
|
Before Width: | Height: | Size: 29 KiB After Width: | Height: | Size: 29 KiB |
|
Before Width: | Height: | Size: 37 KiB After Width: | Height: | Size: 37 KiB |
@@ -0,0 +1,9 @@
|
||||
from shared.docstore.classes import ModelData
|
||||
|
||||
if __name__ == '__main__':
|
||||
# instantiate data object
|
||||
vis_com_list = [ModelData.from_random() for i in range(3)]
|
||||
# generate random predictions
|
||||
[vis_com.from_random() for vis_com in vis_com_list]
|
||||
for vis_com in vis_com_list:
|
||||
print(vis_com)
|
||||
@@ -3,10 +3,10 @@ from __future__ import annotations
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from database import connect
|
||||
from database import total_annotated
|
||||
from dotenv import load_dotenv
|
||||
|
||||
from shared.docstore import connect_mongodb, count_documents
|
||||
|
||||
if __name__ == '__main__':
|
||||
# prepare env vars
|
||||
env_path = Path(__file__).parent.parent / 'local.env'
|
||||
@@ -14,7 +14,10 @@ if __name__ == '__main__':
|
||||
load_dotenv(env_path)
|
||||
os.environ['MONGO_HOST'] = 'localhost'
|
||||
# connect to database
|
||||
collection, db, client = connect()
|
||||
collection, db, client = connect_mongodb()
|
||||
# get visual communication
|
||||
num_docs = total_annotated(collection)
|
||||
num_docs = count_documents(
|
||||
collection=collection,
|
||||
only_with_annotation=True,
|
||||
)
|
||||
print(f"number of annotated documents in database: {num_docs}")
|
||||
@@ -3,10 +3,10 @@ from __future__ import annotations
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from database import connect
|
||||
from database import total_documents
|
||||
from dotenv import load_dotenv
|
||||
|
||||
from shared.docstore import connect_mongodb, count_documents
|
||||
|
||||
if __name__ == '__main__':
|
||||
# prepare env vars
|
||||
env_path = Path(__file__).parent.parent / 'local.env'
|
||||
@@ -14,7 +14,10 @@ if __name__ == '__main__':
|
||||
load_dotenv(env_path)
|
||||
os.environ['MONGO_HOST'] = 'localhost'
|
||||
# connect to database
|
||||
collection, db, client = connect()
|
||||
collection, db, client = connect_mongodb()
|
||||
# get visual communication
|
||||
num_docs = total_documents(collection)
|
||||
num_docs = count_documents(
|
||||
collection=collection,
|
||||
only_with_annotation=False,
|
||||
)
|
||||
print(f"total number of documents in database: {num_docs}")
|
||||
@@ -0,0 +1,16 @@
|
||||
services:
|
||||
vcda_model:
|
||||
image: vcda_model:test
|
||||
container_name: vcda_model_alone
|
||||
build:
|
||||
context: ../
|
||||
dockerfile: ./Dockerfile.model
|
||||
env_file:
|
||||
- ../server.env
|
||||
environment:
|
||||
- ENV=TEST
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
cpus: '2.000'
|
||||
memory: 8G
|
||||
@@ -0,0 +1,14 @@
|
||||
{
|
||||
"datasets": {
|
||||
"train": {
|
||||
"filelist": "datasets/train.csv"
|
||||
},
|
||||
"val": {
|
||||
"filelist": "datasets/val.csv"
|
||||
}
|
||||
},
|
||||
"backbone": {
|
||||
"class": "model.src.models.VisualCommunicationModel",
|
||||
"model_name": "pretrained"
|
||||
}
|
||||
}
|
||||
@@ -1,151 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import random
|
||||
from io import BytesIO
|
||||
|
||||
from torch import Tensor
|
||||
from torch.utils.data import Dataset
|
||||
from torchvision.io import read_image
|
||||
from torchvision.transforms import ColorJitter
|
||||
from torchvision.transforms import InterpolationMode
|
||||
from torchvision.transforms import Normalize
|
||||
from torchvision.transforms.functional import hflip
|
||||
from torchvision.transforms.functional import pad
|
||||
from torchvision.transforms.functional import resize
|
||||
from torchvision.transforms.functional import rotate
|
||||
|
||||
from core.database import connect
|
||||
from core.database import list_names
|
||||
|
||||
# resnet18 original normalization values
|
||||
RESNET_NORMALIZE_MEAN = [0.485, 0.456, 0.406]
|
||||
RESNET_NORMALIZE_STD = [0.229, 0.224, 0.225]
|
||||
|
||||
|
||||
class VCDADataset(Dataset):
|
||||
def __init__(
|
||||
self,
|
||||
do_augment: bool = False,
|
||||
random_annotations: bool = False,
|
||||
normalize_mean: list[float] = RESNET_NORMALIZE_MEAN,
|
||||
normalize_std: list[float] = RESNET_NORMALIZE_STD,
|
||||
):
|
||||
super().__init__()
|
||||
self.do_augment = do_augment
|
||||
self.random_annotations = random_annotations
|
||||
self.normalize_mean = normalize_mean
|
||||
self.normalize_std = normalize_std
|
||||
# prepare augmentation functions
|
||||
self.normalize = Normalize(
|
||||
mean=normalize_mean,
|
||||
std=normalize_std,
|
||||
)
|
||||
self.color_jitter = ColorJitter(
|
||||
brightness=1e-1,
|
||||
contrast=8e-2,
|
||||
saturation=8e-2,
|
||||
)
|
||||
# connect to database
|
||||
collection, _, _ = connect()
|
||||
self.collection = collection
|
||||
# load data names
|
||||
has_annotation = False if self.random_annotations else True
|
||||
self.data_name_list = list_names(
|
||||
collection=self.collection,
|
||||
has_annotation=has_annotation,
|
||||
)
|
||||
|
||||
def __len__(self):
|
||||
return len(self.data_name_list)
|
||||
|
||||
def __getitem__(self, idx):
|
||||
# get image from database
|
||||
name = self.data_name_list[idx]
|
||||
query = {
|
||||
'name': name,
|
||||
}
|
||||
projection = {
|
||||
'_id': False,
|
||||
'image': True,
|
||||
}
|
||||
img_bytes = self.collection.find_one(
|
||||
filter=query,
|
||||
projection=projection,
|
||||
)
|
||||
img = self.load_image(img_bytes)
|
||||
if self.do_augment:
|
||||
img = self.augment(img)
|
||||
return img
|
||||
|
||||
def load_image(
|
||||
self,
|
||||
data: bytes,
|
||||
) -> Tensor:
|
||||
"""Load images tensor from bytes."""
|
||||
assert isinstance(data, bytes)
|
||||
img = read_image(BytesIO(data))
|
||||
img /= 255 # normalize 8-bit image
|
||||
img = self.square_pad(img)
|
||||
img = resize(
|
||||
img=img,
|
||||
size=(512, 512),
|
||||
interpolation=InterpolationMode.BICUBIC,
|
||||
)
|
||||
return img
|
||||
|
||||
@staticmethod
|
||||
def square_pad(
|
||||
img: Tensor,
|
||||
) -> Tensor:
|
||||
"""
|
||||
Pads image to a square with side length
|
||||
equal to the largest side of the input image.
|
||||
"""
|
||||
assert isinstance(img, Tensor)
|
||||
# B, nc, w, h = img.shape
|
||||
h = img.shape[-2]
|
||||
w = img.shape[-1]
|
||||
if h == w:
|
||||
return img
|
||||
max_wh = max([h, w])
|
||||
hp = int((max_wh - w) / 2)
|
||||
vp = int((max_wh - h) / 2)
|
||||
padding = (hp, vp, hp, vp)
|
||||
return pad(img, padding, 0, 'constant')
|
||||
|
||||
def augment(
|
||||
self,
|
||||
img: Tensor,
|
||||
) -> Tensor:
|
||||
"""
|
||||
Augment image with random horizontal flips,
|
||||
rotations and color jitter.
|
||||
"""
|
||||
assert isinstance(img, Tensor)
|
||||
# left-right flip
|
||||
if random.random() >= 0.5:
|
||||
img = hflip(img)
|
||||
# rotation
|
||||
rnd = random.random()
|
||||
if rnd < 0.25:
|
||||
img = rotate(img, angle=90)
|
||||
if rnd < 0.5:
|
||||
img = rotate(img, angle=180)
|
||||
if rnd < 0.75:
|
||||
img = rotate(img, angle=270)
|
||||
# color jitter
|
||||
img = self.color_jitter(img)
|
||||
return img
|
||||
|
||||
def reverse_normalise(
|
||||
self,
|
||||
img: Tensor,
|
||||
) -> Tensor:
|
||||
"""
|
||||
Reverse normalization to get an image
|
||||
that can be interpreted by humans.
|
||||
"""
|
||||
assert isinstance(img, Tensor)
|
||||
img *= Tensor(self.normalize_std).reshape((3, 1, 1))
|
||||
img += Tensor(self.normalize_mean).reshape((3, 1, 1))
|
||||
return img
|
||||
@@ -1,7 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import torch.nn as nn
|
||||
|
||||
|
||||
def setup_criterion():
|
||||
return nn.MSELoss()
|
||||
@@ -0,0 +1,43 @@
|
||||
"""Main script to be run by service."""
|
||||
|
||||
import json
|
||||
import logging
|
||||
from traceback import print_exc
|
||||
|
||||
import torch
|
||||
from models import VisualCommunicationModel
|
||||
from tqdm import tqdm
|
||||
from utils import DEVICE, VCDADataset, load_model
|
||||
|
||||
from shared.utils import setup_logging
|
||||
|
||||
if __name__ == '__main__':
|
||||
# setup logging
|
||||
setup_logging()
|
||||
# instantiate model
|
||||
model: VisualCommunicationModel = load_model()
|
||||
model.eval()
|
||||
# setup dataset
|
||||
dataset = VCDADataset(
|
||||
data_name_list=[
|
||||
'02dbaf48d713e4e6d3a6b98fd2dc866e',
|
||||
],
|
||||
do_augment=False,
|
||||
)
|
||||
|
||||
# make prediction
|
||||
with torch.no_grad():
|
||||
for i in tqdm(range(len(dataset))):
|
||||
try:
|
||||
# get image
|
||||
image = dataset[i]
|
||||
image = torch.unsqueeze(image, 0) # add artificial batch dimension
|
||||
image = image.to(DEVICE)
|
||||
# make prediction
|
||||
pred: dict = model(image)
|
||||
except Exception:
|
||||
print_exc()
|
||||
continue
|
||||
else:
|
||||
print(json.dumps(pred, indent=4))
|
||||
logging.debug('finished')
|
||||
@@ -0,0 +1 @@
|
||||
2bbe2be17a663f4299657378ac5e993e
|
||||
@@ -0,0 +1,3 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from .visual_communication import VisualCommunicationModel
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class AngleTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=3)
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class ContactTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=2)
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class DistanceTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=3)
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class FramingTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=4)
|
||||
@@ -0,0 +1,24 @@
|
||||
from torch import nn
|
||||
|
||||
|
||||
class FullyConnectedModel(nn.Module):
|
||||
"""Fully connected layers model template for intrepreting feature space-
|
||||
output from ResNet18 head."""
|
||||
|
||||
def __init__(self, num_out_features: int):
|
||||
super().__init__()
|
||||
# define layers
|
||||
self.fc1 = nn.Linear(in_features=512, out_features=128)
|
||||
self.af1 = nn.ReLU()
|
||||
self.fc2 = nn.Linear(in_features=128, out_features=32)
|
||||
self.af2 = nn.ReLU()
|
||||
self.fc3 = nn.Linear(in_features=32, out_features=num_out_features)
|
||||
|
||||
def forward(self, x):
|
||||
"""Pass input through model."""
|
||||
x = self.fc1(x)
|
||||
x = self.af1(x)
|
||||
x = self.fc2(x)
|
||||
x = self.af2(x)
|
||||
x = self.fc3(x)
|
||||
return x
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class InformationValueTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=3)
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class ModalityColorTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=3)
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class ModalityDepthTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=3)
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class ModalityLightingTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=3)
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class PointOfViewTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=2)
|
||||
@@ -5,11 +5,15 @@ import torchvision
|
||||
|
||||
|
||||
class ResNet18Head(nn.Module):
|
||||
def __init__(self):
|
||||
def __init__(self, download_resnet_weights: bool = False):
|
||||
super().__init__()
|
||||
# copy out parts from ResNet18 with weights
|
||||
if download_resnet_weights:
|
||||
weights = torchvision.models.ResNet18_Weights.IMAGENET1K_V1
|
||||
else:
|
||||
weights = None
|
||||
resnet18 = torchvision.models.resnet18(
|
||||
weights=torchvision.models.ResNet18_Weights.IMAGENET1K_V1,
|
||||
weights=weights,
|
||||
)
|
||||
# save relevant layers
|
||||
self.conv1 = resnet18.conv1
|
||||
@@ -21,7 +25,7 @@ class ResNet18Head(nn.Module):
|
||||
self.layer3 = resnet18.layer3
|
||||
self.layer4 = resnet18.layer4
|
||||
self.avgpool = resnet18.avgpool
|
||||
self.flat = nn.Flatten() # size 512
|
||||
self.flat = nn.Flatten() # size 512
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv1(x)
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class SalienceTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=5)
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Definition of VisualCommunicationModel class."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from torch import nn
|
||||
|
||||
from .angle import AngleTail
|
||||
from .contact import ContactTail
|
||||
from .distance import DistanceTail
|
||||
from .framing import FramingTail
|
||||
from .information_value import InformationValueTail
|
||||
from .modality_color import ModalityColorTail
|
||||
from .modality_depth import ModalityDepthTail
|
||||
from .modality_lighting import ModalityLightingTail
|
||||
from .point_of_view import PointOfViewTail
|
||||
from .resnet18_head import ResNet18Head
|
||||
from .salience import SalienceTail
|
||||
from .visual_syntax import VisualSyntaxTail
|
||||
|
||||
|
||||
class VisualCommunicationModel(nn.Module):
|
||||
"""Visual communication model."""
|
||||
|
||||
def __init__(self, download_resnet_weights: bool = False):
|
||||
super().__init__()
|
||||
# store other models
|
||||
self.resnet_head = ResNet18Head(download_resnet_weights)
|
||||
self.visual_syntax_tail = VisualSyntaxTail()
|
||||
self.contact_tail = ContactTail()
|
||||
self.angle_tail = AngleTail()
|
||||
self.point_of_view_tail = PointOfViewTail()
|
||||
self.distance_tail = DistanceTail()
|
||||
self.modality_lighting_tail = ModalityLightingTail()
|
||||
self.modality_color_tail = ModalityColorTail()
|
||||
self.modality_depth_tail = ModalityDepthTail()
|
||||
self.information_value_tail = InformationValueTail()
|
||||
self.framing_tail = FramingTail()
|
||||
self.salience_tail = SalienceTail()
|
||||
|
||||
def forward(self, x) -> dict:
|
||||
"""Calculate model output on data."""
|
||||
# generate visual representation
|
||||
features = self.resnet_head(x)
|
||||
# make predictions
|
||||
prediction_dict = {
|
||||
'visual_syntax': self.visual_syntax_tail(features),
|
||||
'contact': self.contact_tail(features),
|
||||
'angle': self.angle_tail(features),
|
||||
'point_of_view': self.point_of_view_tail(features),
|
||||
'distance': self.distance_tail(features),
|
||||
'modality_lighting': self.modality_lighting_tail(features),
|
||||
'modality_color': self.modality_color_tail(features),
|
||||
'modality_depth': self.modality_depth_tail(features),
|
||||
'information_value': self.information_value_tail(features),
|
||||
'framing': self.framing_tail(features),
|
||||
'salience': self.salience_tail(features),
|
||||
}
|
||||
return prediction_dict
|
||||
@@ -0,0 +1,6 @@
|
||||
from .fully_connected import FullyConnectedModel
|
||||
|
||||
|
||||
class VisualSyntaxTail(FullyConnectedModel):
|
||||
def __init__(self):
|
||||
super().__init__(num_out_features=18)
|
||||
@@ -1,6 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from dataloader import VCDADataset # noqa: F401
|
||||
from loss_fn import setup_criterion # noqa: F401
|
||||
|
||||
from model import ResNet18Head # noqa: F401
|
||||
@@ -0,0 +1,3 @@
|
||||
from .get_class import get_class
|
||||
from .load_model import DEVICE, load_model
|
||||
from .vcda_dataset import VCDADataset
|
||||
@@ -0,0 +1,11 @@
|
||||
from importlib import import_module
|
||||
|
||||
from torch.nn import Module
|
||||
|
||||
|
||||
def get_class(path: str) -> Module:
|
||||
parts = path.split('.')
|
||||
module_path = '.'.join(parts[:-1])
|
||||
class_name = parts[-1]
|
||||
module = import_module(module_path)
|
||||
return getattr(module, class_name)
|
||||
@@ -0,0 +1,19 @@
|
||||
"""Definition of get_model_name function."""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def get_model_name(
|
||||
path: Path,
|
||||
) -> str:
|
||||
"""Get model name from model_name file."""
|
||||
assert isinstance(path, Path)
|
||||
assert path.exists(), f'{path} does not exist'
|
||||
# read file contents
|
||||
with open(file=path, encoding='utf-8') as fh:
|
||||
model_name = fh.read()
|
||||
# remove newline character
|
||||
model_name = model_name.strip()
|
||||
logging.debug('finished')
|
||||
return model_name
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Definition of load_model function."""
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
|
||||
from model.src.models import VisualCommunicationModel
|
||||
from shared.repositories import ModelRepository
|
||||
|
||||
from .get_model_name import get_model_name
|
||||
|
||||
DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
|
||||
|
||||
|
||||
def load_model() -> VisualCommunicationModel:
|
||||
"""Instantiate model with weights loaded from latest model saved in
|
||||
MinIO."""
|
||||
# instantiate model
|
||||
model = VisualCommunicationModel()
|
||||
# get model object name
|
||||
model_name_path = Path('model_name.txt')
|
||||
model_object_name = get_model_name(path=model_name_path)
|
||||
logging.info('using model: %s', model_object_name)
|
||||
# load model data
|
||||
with ModelRepository() as repo:
|
||||
model_data = repo.get_data(model_object_name)
|
||||
if model_data is None:
|
||||
raise FileNotFoundError(f'model {model_object_name} not found')
|
||||
model_checkpoint = torch.load(model_data.buffer)
|
||||
model.load_state_dict(model_checkpoint)
|
||||
# clean memory
|
||||
model_checkpoint.clear()
|
||||
# move model to selected device
|
||||
model = model.to(DEVICE)
|
||||
logging.debug('finished')
|
||||
return model
|
||||
@@ -0,0 +1,49 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
from dataloader import VCDADataset # noqa: F401
|
||||
|
||||
from ..models import VisualCommunicationModel
|
||||
|
||||
PRE_WARMUP_LR = 1e-10
|
||||
POST_WARMUP_LR = 1e-5
|
||||
BATCH_SIZE = 32
|
||||
MODEL_NAME = 'visual_communication_model_v1'
|
||||
MAX_EPOCHS = 200
|
||||
RUN_NAME = datetime.now().strftime('%Y-%m-%d-%H%M') + '_' + MODEL_NAME
|
||||
DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
|
||||
|
||||
|
||||
# create loss function
|
||||
criterion = nn.MSELoss()
|
||||
|
||||
|
||||
def loss_fn(pred, target):
|
||||
"""Create loss function."""
|
||||
return criterion(pred, target.unsqueeze(-1))
|
||||
|
||||
|
||||
# create model
|
||||
model = VisualCommunicationModel().to(DEVICE)
|
||||
|
||||
# create optimizer
|
||||
optim = torch.optim.Adam(model.parameters(), lr=POST_WARMUP_LR)
|
||||
|
||||
# create datasets
|
||||
# train_dataset
|
||||
# validation_dataset
|
||||
|
||||
# create dataloaders
|
||||
|
||||
# create trainer and evaluator
|
||||
|
||||
# setup progressbar
|
||||
|
||||
# setup lr scheduler
|
||||
|
||||
# setup checkpoint saving
|
||||
|
||||
# load in checkpoint if exists
|
||||
@@ -0,0 +1,134 @@
|
||||
"""Definition of Visual Critial Discourse Analysis dataloader."""
|
||||
|
||||
import random
|
||||
|
||||
from PIL import Image
|
||||
from torch import Tensor
|
||||
from torch.utils.data import Dataset
|
||||
from torchvision.transforms import ColorJitter, InterpolationMode, Normalize
|
||||
from torchvision.transforms.functional import (
|
||||
hflip,
|
||||
pad,
|
||||
resize,
|
||||
rotate,
|
||||
to_pil_image,
|
||||
to_tensor,
|
||||
)
|
||||
|
||||
from shared.repositories import ImageRepository
|
||||
|
||||
# resnet18 original normalization values
|
||||
RESNET_NORMALIZE_MEAN = [0.485, 0.456, 0.406]
|
||||
RESNET_NORMALIZE_STD = [0.229, 0.224, 0.225]
|
||||
|
||||
|
||||
class VCDADataset(Dataset):
|
||||
"""VCDA dataset class."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
data_name_list: list[str],
|
||||
do_augment: bool = False,
|
||||
random_annotations: bool = False,
|
||||
):
|
||||
super().__init__()
|
||||
self.data_name_list = data_name_list
|
||||
self.do_augment = do_augment
|
||||
self.random_annotations = random_annotations
|
||||
self.normalize_mean = RESNET_NORMALIZE_MEAN
|
||||
self.normalize_std = RESNET_NORMALIZE_STD
|
||||
# prepare augmentation functions
|
||||
self.normalize = Normalize(
|
||||
mean=self.normalize_mean,
|
||||
std=self.normalize_std,
|
||||
)
|
||||
self.color_jitter = ColorJitter(
|
||||
brightness=1e-1,
|
||||
contrast=8e-2,
|
||||
saturation=8e-2,
|
||||
)
|
||||
|
||||
def __len__(self):
|
||||
return len(self.data_name_list)
|
||||
|
||||
def __getitem__(self, idx):
|
||||
# get image from database
|
||||
image_name = self.data_name_list[idx]
|
||||
with ImageRepository() as repo:
|
||||
image_data = repo.get_data(image_name)
|
||||
assert image_data is not None
|
||||
tensor = self.image_to_tensor(image_data.image)
|
||||
if self.do_augment:
|
||||
tensor = self.augment(tensor)
|
||||
return tensor
|
||||
|
||||
def image_to_tensor(
|
||||
self,
|
||||
image: Image.Image,
|
||||
) -> Tensor:
|
||||
"""Load images tensor from bytes."""
|
||||
tensor = to_tensor(image)
|
||||
tensor /= 255 # normalize 8-bit image
|
||||
tensor = self.square_pad(tensor=tensor)
|
||||
tensor = resize(
|
||||
img=tensor,
|
||||
size=(512, 512),
|
||||
interpolation=InterpolationMode.BICUBIC,
|
||||
)
|
||||
return tensor
|
||||
|
||||
@staticmethod
|
||||
def square_pad(
|
||||
tensor: Tensor,
|
||||
) -> Tensor:
|
||||
"""Pads image to a square with side length equal to the largest side of
|
||||
the input image."""
|
||||
assert isinstance(tensor, Tensor)
|
||||
# B, nc, w, h = img.shape
|
||||
h = tensor.shape[-2]
|
||||
w = tensor.shape[-1]
|
||||
if h == w:
|
||||
return tensor
|
||||
max_wh = max([h, w])
|
||||
hp = int((max_wh - w) / 2)
|
||||
vp = int((max_wh - h) / 2)
|
||||
padding = (hp, vp, hp, vp)
|
||||
tensor = pad(tensor, padding, 0, 'constant')
|
||||
return tensor
|
||||
|
||||
def augment(
|
||||
self,
|
||||
tensor: Tensor,
|
||||
) -> Tensor:
|
||||
"""Augment image with random horizontal flips, rotations and color
|
||||
jitter."""
|
||||
assert isinstance(tensor, Tensor)
|
||||
# left-right flip
|
||||
if random.random() >= 0.5:
|
||||
tensor = hflip(tensor)
|
||||
# rotation
|
||||
rnd = random.random()
|
||||
if rnd < 0.25:
|
||||
tensor = rotate(tensor, angle=90)
|
||||
if rnd < 0.5:
|
||||
tensor = rotate(tensor, angle=180)
|
||||
if rnd < 0.75:
|
||||
tensor = rotate(tensor, angle=270)
|
||||
# color jitter
|
||||
tensor = self.color_jitter(tensor)
|
||||
return tensor
|
||||
|
||||
def reverse_normalise(
|
||||
self,
|
||||
tensor: Tensor,
|
||||
) -> Image.Image:
|
||||
"""Reverse normalization to get an image that can be interpreted by
|
||||
humans."""
|
||||
assert isinstance(tensor, Tensor)
|
||||
tensor *= Tensor(self.normalize_std).reshape((3, 1, 1))
|
||||
tensor += Tensor(self.normalize_mean).reshape((3, 1, 1))
|
||||
image = to_pil_image(
|
||||
pic=tensor,
|
||||
mode='RGB',
|
||||
)
|
||||
return image
|
||||
+163
@@ -0,0 +1,163 @@
|
||||
import json
|
||||
import os
|
||||
from argparse import ArgumentParser
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
from dotenv import load_dotenv
|
||||
from ignite.engine import Events, create_supervised_evaluator, create_supervised_trainer
|
||||
from ignite.handlers import global_step_from_engine
|
||||
from ignite.handlers.checkpoint import Checkpoint
|
||||
from ignite.handlers.param_scheduler import create_lr_scheduler_with_warmup
|
||||
from ignite.handlers.tensorboard_logger import TensorboardLogger
|
||||
from ignite.handlers.tqdm_logger import ProgressBar
|
||||
from ignite.metrics import Average, Loss, RunningAverage
|
||||
from torch import nn
|
||||
from torch.optim.lr_scheduler import ExponentialLR
|
||||
from torch.utils.data import DataLoader
|
||||
|
||||
from model.src.utils import VCDADataset, get_class
|
||||
|
||||
|
||||
def parse_arguments():
|
||||
parser = ArgumentParser()
|
||||
parser.add_argument('-c', '--config', type=Path, required=True)
|
||||
parser.add_argument('-d', '--device', type=str, default='cpu')
|
||||
parser.add_argument('-b', '--batch-size', type=int, default=32)
|
||||
parser.add_argument('--epochs', type=int, default=50)
|
||||
parser.add_argument('--epoch-length', type=int, default=128)
|
||||
parser.add_argument('--checkpoint', type=Path)
|
||||
parser.add_argument('--lr', type=float, default=1e-5)
|
||||
parser.add_argument('--lr-gamma', type=float, default=0.95)
|
||||
parser.add_argument('--lr-warmup-start', type=float, default=1e-8)
|
||||
parser.add_argument('--lr-warmup-duration', type=int, default=5)
|
||||
parser.add_argument('--train-seed', type=int, default=1312)
|
||||
parser.add_argument('--val-seed', type=int, default=1313)
|
||||
parser.add_argument('--loader-workers', type=int, default=8)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
args = parse_arguments()
|
||||
|
||||
load_dotenv('server.env')
|
||||
|
||||
# read model config
|
||||
with open(args.config) as fh:
|
||||
config = json.load(fh)
|
||||
model_name = args.config.stem
|
||||
run_name = datetime.now(timezone.utc).strftime('%Y-%m-%d-%H%M') + '_' + model_name
|
||||
device = torch.device(args.device)
|
||||
|
||||
# create model
|
||||
model_class = get_class(config['backbone']['class'])
|
||||
model = model_class().to(device)
|
||||
optimizer = torch.optim.Adam(model.parameters(), lr=args.lr)
|
||||
loss_fn = nn.CrossEntropyLoss()
|
||||
|
||||
# create datasets and loaders
|
||||
with open('model/src/dataset/train.csv', encoding='utf-8') as fh:
|
||||
train_data_name_list = fh.read().split('\n')
|
||||
train_dataset = VCDADataset(
|
||||
data_name_list=train_data_name_list,
|
||||
)
|
||||
train_loader = DataLoader(dataset=train_dataset, num_workers=args.loader_workers)
|
||||
|
||||
with open('model/src/dataset/val.csv', encoding='utf-8') as fh:
|
||||
val_data_name_list = fh.read().split('\n')
|
||||
val_dataset = VCDADataset(data_name_list=val_data_name_list)
|
||||
val_loader = DataLoader(dataset=val_dataset, num_workers=args.loader_workers)
|
||||
|
||||
# create trainer and evaluator
|
||||
trainer = create_supervised_trainer(
|
||||
model=model,
|
||||
optimizer=optimizer,
|
||||
loss_fn=loss_fn,
|
||||
device=device,
|
||||
)
|
||||
Average(output_transform=lambda x: x).attach(trainer, name='loss')
|
||||
RunningAverage(output_transform=lambda x: x, alpha=0.5).attach(
|
||||
trainer,
|
||||
'running_avg_loss',
|
||||
)
|
||||
ProgressBar(desc='Train', ncols=80).attach(trainer, ['running_avg_loss'])
|
||||
|
||||
val_metrics = {
|
||||
'loss': Loss(loss_fn=loss_fn, device=device),
|
||||
}
|
||||
evaluator = create_supervised_evaluator(
|
||||
model=model,
|
||||
metrics=val_metrics, # type: ignore
|
||||
device=device,
|
||||
)
|
||||
ProgressBar(desc='Val', ncols=80).attach(evaluator)
|
||||
|
||||
# create lr scheduler
|
||||
lr_scheduler = ExponentialLR(
|
||||
optimizer=optimizer,
|
||||
gamma=args.lr_gamma,
|
||||
)
|
||||
lr_handler = create_lr_scheduler_with_warmup(
|
||||
lr_scheduler=lr_scheduler,
|
||||
warmup_start_value=args.lr_warmup_start,
|
||||
warmup_duration=args.lr_warmup_duration,
|
||||
warmup_end_value=args.lr,
|
||||
)
|
||||
trainer.add_event_handler(Events.EPOCH_STARTED, lr_handler)
|
||||
|
||||
|
||||
# log metrics to TensorBoard
|
||||
@trainer.on(Events.EPOCH_COMPLETED)
|
||||
def evaluate():
|
||||
evaluator.run(val_loader, epoch_length=args.val_epoch_length)
|
||||
|
||||
|
||||
tb_logger = TensorboardLogger(log_dir=f'runs/logs/{run_name}')
|
||||
tb_logger.attach_opt_params_handler(
|
||||
engine=trainer,
|
||||
event_name=Events.EPOCH_COMPLETED,
|
||||
optimizer=optimizer,
|
||||
)
|
||||
for tag, engine in [('train', trainer), ('val', evaluator)]:
|
||||
tb_logger.attach_output_handler(
|
||||
engine,
|
||||
event_name=Events.EPOCH_COMPLETED,
|
||||
tag=tag,
|
||||
metric_names='all',
|
||||
global_step_transform=global_step_from_engine(trainer),
|
||||
)
|
||||
|
||||
# Set up checkpoint saving
|
||||
to_save = {
|
||||
'model': model,
|
||||
'optimizer': optimizer,
|
||||
'trainer': trainer,
|
||||
'lr_scheduler': lr_scheduler,
|
||||
'lr_handler': lr_handler,
|
||||
}
|
||||
checkpoint_handler = Checkpoint(
|
||||
to_save,
|
||||
f'runs/checkpoints/{run_name}',
|
||||
n_saved=3,
|
||||
filename_prefix='best',
|
||||
score_function=lambda engine: -engine.state.metrics['loss'],
|
||||
score_name='neg_val_loss',
|
||||
global_step_transform=global_step_from_engine(trainer),
|
||||
)
|
||||
evaluator.add_event_handler(Events.COMPLETED, checkpoint_handler)
|
||||
|
||||
if args.checkpoint:
|
||||
Checkpoint.load_objects(to_load=to_save, checkpoint=str(args.checkpoint))
|
||||
|
||||
# save model config
|
||||
os.makedirs('runs/configs/', exist_ok=True)
|
||||
with open(f'runs/configs/{run_name}.json', 'w', encoding='utf-8') as fh:
|
||||
json.dump(config, fh)
|
||||
|
||||
# start training
|
||||
trainer.run(
|
||||
train_loader,
|
||||
max_epochs=args.epochs,
|
||||
epoch_length=args.epoch_length,
|
||||
)
|
||||
tb_logger.close()
|
||||
@@ -0,0 +1,124 @@
|
||||
"""Script to move all images from MongoDB to MinIO."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
from bson import ObjectId
|
||||
from dotenv import load_dotenv
|
||||
from pymongo.collection import Collection
|
||||
|
||||
from shared.datastore import Datastore
|
||||
from shared.docstore import connect_mongodb
|
||||
from shared.docstore.src.classes import VisualCommunication
|
||||
from shared.utils import check_env, setup_logging
|
||||
from web_ui.src.main import NECESSARY_ENV_VAR_LIST
|
||||
|
||||
|
||||
def list_mongo_document_ids(
|
||||
collection: Collection,
|
||||
) -> list[ObjectId]:
|
||||
"""Get list of all VisualCommunication documents in MongoDB."""
|
||||
# prepare query
|
||||
query = collection.find(
|
||||
filter={},
|
||||
projection={'_id': True},
|
||||
)
|
||||
# execute query
|
||||
res_list = list(query)
|
||||
logging.debug('got %s document(s)', len(res_list))
|
||||
# convert result
|
||||
id_list = [elem['_id'] for elem in res_list]
|
||||
return id_list
|
||||
|
||||
|
||||
def get_visual_communication(
|
||||
collection: Collection,
|
||||
doc_id: ObjectId,
|
||||
) -> VisualCommunication:
|
||||
"""Get image from MongoDB."""
|
||||
# prepare query
|
||||
query = collection.find(
|
||||
filter={'_id': doc_id},
|
||||
projection={'_id': False},
|
||||
)
|
||||
# execute query
|
||||
res_list = list(query)
|
||||
logging.debug('got %s document(s)', len(res_list))
|
||||
# convert result
|
||||
vis_com = VisualCommunication.model_validate(res_list[0])
|
||||
return vis_com
|
||||
|
||||
|
||||
def update_visual_communication(
|
||||
collection: Collection,
|
||||
vis_com: VisualCommunication,
|
||||
) -> None:
|
||||
"""Update VisualCommunication document in MongoDB."""
|
||||
try:
|
||||
collection.update_one(
|
||||
filter={
|
||||
'name': vis_com.name,
|
||||
},
|
||||
update={
|
||||
'$set': vis_com.model_dump(),
|
||||
},
|
||||
)
|
||||
except Exception as exc:
|
||||
logging.error('failed saving %s', vis_com.name)
|
||||
logging.debug(exc)
|
||||
raise exc
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
# load in env file
|
||||
env_path = Path(__file__).parent.parent / 'server.env'
|
||||
assert env_path.exists()
|
||||
load_dotenv(env_path)
|
||||
# ensure env vars set
|
||||
check_env(NECESSARY_ENV_VAR_LIST)
|
||||
# setup logging
|
||||
setup_logging()
|
||||
# connect to minIO
|
||||
datastore = Datastore()
|
||||
datastore.connect()
|
||||
assert datastore._client is not None
|
||||
# connect to MongoDB
|
||||
collection, db, client = connect_mongodb()
|
||||
# list documents in mongoDB
|
||||
id_list = list_mongo_document_ids(collection)
|
||||
for doc_id in id_list:
|
||||
try:
|
||||
# get image from MongoDB
|
||||
vis_com = get_visual_communication(collection, doc_id)
|
||||
except Exception as exc:
|
||||
logging.debug(exc)
|
||||
logging.error('failed getting image from document: %s', doc_id)
|
||||
continue
|
||||
try:
|
||||
# get image
|
||||
image = vis_com.get_image(
|
||||
minio_client=datastore._client,
|
||||
)
|
||||
# put buffer in minio
|
||||
object_name = datastore.put_image(
|
||||
image=image,
|
||||
)
|
||||
except Exception as exc:
|
||||
logging.debug(exc)
|
||||
logging.error('failed saving image to minio')
|
||||
continue
|
||||
# update visual communication in mongodb
|
||||
try:
|
||||
vis_com.image = None # type: ignore
|
||||
vis_com.object_name = object_name
|
||||
update_visual_communication(collection, vis_com)
|
||||
except Exception as exc:
|
||||
logging.debug(exc)
|
||||
logging.error(
|
||||
'failed updating visual communication %s',
|
||||
vis_com.name,
|
||||
)
|
||||
continue
|
||||
logging.debug('updated visual communication %s', vis_com.name)
|
||||
Generated
+2523
-767
File diff suppressed because it is too large
Load Diff
+92
-11
@@ -5,26 +5,26 @@ description = ""
|
||||
authors = ["Brian Bjarke Jensen <bbj@skov.dk>"]
|
||||
readme = "README.md"
|
||||
packages = [
|
||||
{ include = "src" },
|
||||
{ include = "shared" },
|
||||
]
|
||||
|
||||
[tool.poetry.dependencies]
|
||||
python = "^3.12"
|
||||
gunicorn = "^21.2.0"
|
||||
python-dotenv = "^1.0.1"
|
||||
dash = "^2.15.0"
|
||||
dash-bootstrap-components = "^1.5.0"
|
||||
dash-mantine-components = "^0.12.1"
|
||||
pydantic = "^2.6.1"
|
||||
pillow = "^10.2.0"
|
||||
pymongo = "^4.6.1"
|
||||
dash-auth = "^2.2.0"
|
||||
|
||||
|
||||
[tool.poetry.group.test.dependencies]
|
||||
flake8 = "^7.0.0"
|
||||
mypy = "^1.8.0"
|
||||
types-pillow = "^10.2.0.20240213"
|
||||
types-requests = "^2.32.0.20240602"
|
||||
types-retry = "^0.9.9.4"
|
||||
flake8-pyproject = "^1.2.3"
|
||||
pandas-stubs = "^2.2.2.240603"
|
||||
types-tqdm = "^4.66.0.20240417"
|
||||
pytest = "^8.3.3"
|
||||
testcontainers = "^4.8.2"
|
||||
coverage = "^7.6.4"
|
||||
pytest-cov = "^6.0.0"
|
||||
|
||||
|
||||
[tool.poetry.group.dev.dependencies]
|
||||
@@ -32,12 +32,93 @@ pandas = "^2.2.1"
|
||||
selenium = "^4.18.1"
|
||||
webdriver-manager = "^4.0.1"
|
||||
retry = "^0.9.2"
|
||||
pre-commit = "^4.1.0"
|
||||
|
||||
|
||||
[tool.poetry.group.model.dependencies]
|
||||
torch = "^2.2.1"
|
||||
torch = "^2.0.0"
|
||||
torchvision = "^0.17.1"
|
||||
torchinfo = "^1.8.0"
|
||||
minio = "^7.2.7"
|
||||
tqdm = "^4.66.4"
|
||||
pytorch-ignite = "^0.5.1"
|
||||
tensorboard = "^2.17.1"
|
||||
|
||||
|
||||
[tool.poetry.group.shared.dependencies]
|
||||
pymongo = "^4.7.2"
|
||||
pydantic = "^2.6.1"
|
||||
pillow = "^10.2.0"
|
||||
|
||||
|
||||
[tool.poetry.group.web_ui.dependencies]
|
||||
dash = "^2.17.0"
|
||||
gunicorn = "^21.2.0"
|
||||
dash-bootstrap-components = "^1.5.0"
|
||||
dash-mantine-components = "^0.12.1"
|
||||
dash-auth = "^2.2.0"
|
||||
|
||||
|
||||
[[tool.poetry.source]]
|
||||
name = "threadripper"
|
||||
url = "http://192.168.1.2:5001/index/"
|
||||
priority = "primary"
|
||||
|
||||
[build-system]
|
||||
requires = ["poetry-core"]
|
||||
build-backend = "poetry.core.masonry.api"
|
||||
|
||||
[tool.isort]
|
||||
profile = "black"
|
||||
|
||||
[tool.flake8]
|
||||
per-file-ignores = "__init__.py:F401"
|
||||
max-line-length = 88
|
||||
extend-ignore = "E203"
|
||||
|
||||
[tool.mypy]
|
||||
exclude = "image_download"
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "dash.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "dash_auth.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "dash_mantine_components.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "dash_bootstrap_components.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "torchvision.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "image_download.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "dataloader.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "utils.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "shared.datastore.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "models.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = "shared.docstore.*"
|
||||
ignore_missing_imports = true
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
from .src import ImageRepository, ModelRepository, VisualCommunicationRepository
|
||||
from .src.dto import (
|
||||
HexadecimalString,
|
||||
ImageData,
|
||||
ModelData,
|
||||
VisualCommunicationData,
|
||||
VisualCommunicationValues,
|
||||
)
|
||||
@@ -0,0 +1,3 @@
|
||||
from .image_repository import ImageRepository
|
||||
from .model_repository import ModelRepository
|
||||
from .visual_communication_repository import VisualCommunicationRepository
|
||||
@@ -0,0 +1,7 @@
|
||||
from .hexadecimal_string import HexadecimalString
|
||||
from .image_data import ImageData
|
||||
from .model_data import ModelData
|
||||
from .visual_communication_data import (
|
||||
VisualCommunicationData,
|
||||
VisualCommunicationValues,
|
||||
)
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Definition of BytesIO Pydantic Annotation."""
|
||||
|
||||
from io import BytesIO
|
||||
from typing import Any
|
||||
|
||||
from pydantic.json_schema import JsonSchemaValue
|
||||
from pydantic_core import core_schema
|
||||
|
||||
|
||||
class BytesIOPydanticAnnotation:
|
||||
"""Pydantic annotation that defines input validation, as well as general
|
||||
and json serialization."""
|
||||
|
||||
@classmethod
|
||||
def validate_input(cls, v: Any, handler) -> BytesIO:
|
||||
"""Pydantic-related function to validate input on instantiation."""
|
||||
if isinstance(v, BytesIO):
|
||||
return v
|
||||
s = handler(v)
|
||||
return BytesIO(s)
|
||||
|
||||
@classmethod
|
||||
def __get_pydantic_core_schema__(
|
||||
cls,
|
||||
source_type,
|
||||
_handler,
|
||||
) -> core_schema.CoreSchema:
|
||||
assert source_type is BytesIO
|
||||
return core_schema.no_info_wrap_validator_function(
|
||||
function=cls.validate_input,
|
||||
schema=core_schema.str_schema(),
|
||||
serialization=core_schema.to_string_ser_schema(),
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def __get_pydantic_json_schema__(cls, _core_schema, handler) -> JsonSchemaValue:
|
||||
return handler(core_schema.str_schema())
|
||||
@@ -0,0 +1,24 @@
|
||||
"""Definition of Checksum DTO."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
|
||||
class HexadecimalString(str):
|
||||
"""Hexadecimal-string class."""
|
||||
|
||||
def __new__(cls, string):
|
||||
# ensure proper input format
|
||||
pattern = r'[0-9-a-fA-F]{32}'
|
||||
match = re.match(pattern, string)
|
||||
if match is None:
|
||||
raise ValueError(f'format does not match a hexadecimal-string: {string}')
|
||||
return super().__new__(cls, string)
|
||||
|
||||
def __repr__(self) -> str:
|
||||
class_name = self.__class__.__name__
|
||||
return f"{class_name}('{self}')"
|
||||
|
||||
def __reduce__(self):
|
||||
return self.__class__, (self,)
|
||||
@@ -0,0 +1,38 @@
|
||||
"""Definition of HexadecimalString Pydantic Annotation."""
|
||||
|
||||
from typing import Any
|
||||
|
||||
from pydantic.json_schema import JsonSchemaValue
|
||||
from pydantic_core import core_schema
|
||||
|
||||
from .hexadecimal_string import HexadecimalString
|
||||
|
||||
|
||||
class HexadecimalStringPydanticAnnotation:
|
||||
"""Pydantic annotation that defines input validation, as well as general
|
||||
and json serialization."""
|
||||
|
||||
@classmethod
|
||||
def validate_input(cls, v: Any, handler) -> HexadecimalString:
|
||||
"""Pydantic-related function to validate input on instantiation."""
|
||||
if isinstance(v, HexadecimalString):
|
||||
return v
|
||||
s = handler(v)
|
||||
return HexadecimalString(s)
|
||||
|
||||
@classmethod
|
||||
def __get_pydantic_core_schema__(
|
||||
cls,
|
||||
source_type,
|
||||
_handler,
|
||||
) -> core_schema.CoreSchema:
|
||||
assert source_type is HexadecimalString
|
||||
return core_schema.no_info_wrap_validator_function(
|
||||
function=cls.validate_input,
|
||||
schema=core_schema.str_schema(),
|
||||
serialization=core_schema.to_string_ser_schema(),
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def __get_pydantic_json_schema__(cls, _core_schema, handler) -> JsonSchemaValue:
|
||||
return handler(core_schema.str_schema())
|
||||
@@ -0,0 +1,16 @@
|
||||
"""Definition of VisualData DTO."""
|
||||
|
||||
from typing import Annotated
|
||||
|
||||
from PIL import Image
|
||||
from pydantic import Field
|
||||
|
||||
from .image_pydantic_annotation import ImagePydanticAnnotation
|
||||
from .type_checking_base_model import TypeCheckingBaseModel
|
||||
|
||||
|
||||
class ImageData(TypeCheckingBaseModel):
|
||||
"""Visual data class."""
|
||||
|
||||
image: Annotated[Image.Image, ImagePydanticAnnotation]
|
||||
name: str = Field(min_length=1)
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Definition of BytesIO Pydantic Annotation."""
|
||||
|
||||
from typing import Any
|
||||
|
||||
from PIL import Image
|
||||
from pydantic.json_schema import JsonSchemaValue
|
||||
from pydantic_core import core_schema
|
||||
|
||||
|
||||
class ImagePydanticAnnotation:
|
||||
"""Pydantic annotation that defines input validation, as well as general
|
||||
and json serialization."""
|
||||
|
||||
@classmethod
|
||||
def validate_input(cls, v: Any, handler) -> Image.Image:
|
||||
"""Pydantic-related function to validate input on instantiation."""
|
||||
if isinstance(v, Image.Image):
|
||||
return v
|
||||
s = handler(v)
|
||||
return Image.open(s)
|
||||
|
||||
@classmethod
|
||||
def __get_pydantic_core_schema__(
|
||||
cls,
|
||||
source_type,
|
||||
_handler,
|
||||
) -> core_schema.CoreSchema:
|
||||
assert source_type is Image.Image
|
||||
return core_schema.no_info_wrap_validator_function(
|
||||
function=cls.validate_input,
|
||||
schema=core_schema.str_schema(),
|
||||
serialization=core_schema.to_string_ser_schema(),
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def __get_pydantic_json_schema__(cls, _core_schema, handler) -> JsonSchemaValue:
|
||||
return handler(core_schema.str_schema())
|
||||
@@ -0,0 +1,60 @@
|
||||
"""Definition of ModelData DTO."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from hashlib import md5
|
||||
from io import BytesIO
|
||||
from typing import Annotated
|
||||
|
||||
import torch
|
||||
from pydantic import Field
|
||||
|
||||
from .bytes_io_pydantic_annotation import BytesIOPydanticAnnotation
|
||||
from .hexadecimal_string import HexadecimalString
|
||||
from .hexadecimal_string_pydantic_annotation import HexadecimalStringPydanticAnnotation
|
||||
from .type_checking_base_model import TypeCheckingBaseModel
|
||||
|
||||
|
||||
class ModelData(TypeCheckingBaseModel):
|
||||
"""Model Data DTO."""
|
||||
|
||||
buffer: Annotated[BytesIO, BytesIOPydanticAnnotation]
|
||||
buffer_checksum: Annotated[HexadecimalString, HexadecimalStringPydanticAnnotation]
|
||||
class_name: str = Field(
|
||||
min_length=1,
|
||||
description='name of model class to generate data.',
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def calculate_checksum(buffer: BytesIO) -> HexadecimalString:
|
||||
"""Calculate buffer checksum."""
|
||||
checksum = md5(buffer.getbuffer()).hexdigest()
|
||||
return HexadecimalString(checksum)
|
||||
|
||||
@staticmethod
|
||||
def model_to_buffer(model: torch.nn.Module) -> BytesIO:
|
||||
"""Save model to buffer."""
|
||||
assert isinstance(model, torch.nn.Module)
|
||||
buffer = BytesIO()
|
||||
torch.save(model.state_dict(), buffer)
|
||||
return buffer
|
||||
|
||||
@classmethod
|
||||
def from_model(cls, model: torch.nn.Module) -> ModelData:
|
||||
"""Instantiate from torch module."""
|
||||
assert isinstance(model, torch.nn.Module)
|
||||
# get model name
|
||||
class_name = type(model).__name__
|
||||
# save data to buffer
|
||||
buffer = cls.model_to_buffer(model)
|
||||
buffer = BytesIO()
|
||||
torch.save(model.state_dict(), buffer)
|
||||
# calculate checksum
|
||||
buffer_checksum = cls.calculate_checksum(buffer)
|
||||
# instantiate from buffer
|
||||
data = cls(
|
||||
buffer=buffer,
|
||||
buffer_checksum=buffer_checksum,
|
||||
class_name=class_name,
|
||||
)
|
||||
return data
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Definition of TypeCheckingBaseModel class."""
|
||||
|
||||
from pydantic import BaseModel, ConfigDict
|
||||
|
||||
|
||||
class TypeCheckingBaseModel(BaseModel):
|
||||
"""BaseModel with added type checking on input types."""
|
||||
|
||||
model_config = ConfigDict(
|
||||
frozen=True, # ensure data immutability
|
||||
)
|
||||
@@ -0,0 +1,2 @@
|
||||
from .visual_communication_data import VisualCommunicationData
|
||||
from .visual_communication_values import VisualCommunicationValues
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Definition of AngleValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class AngleValues(ValuesModel):
|
||||
"""Angle values DTO."""
|
||||
|
||||
high: float
|
||||
eye_level: float
|
||||
low: float
|
||||
@@ -0,0 +1,10 @@
|
||||
"""Definition of ContactValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class ContactValues(ValuesModel):
|
||||
"""Contact values DTO."""
|
||||
|
||||
offer: float
|
||||
demand: float
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Definition of DistanceValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class DistanceValues(ValuesModel):
|
||||
"""Distance values DTO."""
|
||||
|
||||
long: float
|
||||
medium: float
|
||||
close: float
|
||||
@@ -0,0 +1,12 @@
|
||||
"""Definition of FramingValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class FramingValues(ValuesModel):
|
||||
"""Framing values DTO."""
|
||||
|
||||
frame_lines: float
|
||||
empty_space: float
|
||||
colour_contrast: float
|
||||
form_contrast: float
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Definition of InformationValueValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class InformationValueValues(ValuesModel):
|
||||
"""Information value values DTO."""
|
||||
|
||||
given_new: float
|
||||
ideal_real: float
|
||||
central_marginal: float
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Definition of ModalityColorValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class ModalityColorValues(ValuesModel):
|
||||
"""Modality color values DTO."""
|
||||
|
||||
high: float
|
||||
medium: float
|
||||
low: float
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Definition of ModalityDepthValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class ModalityDepthValues(ValuesModel):
|
||||
"""Modality depth values DTO."""
|
||||
|
||||
high: float
|
||||
medium: float
|
||||
low: float
|
||||
@@ -0,0 +1,11 @@
|
||||
"""Definition of ModalityLightingValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class ModalityLightingValues(ValuesModel):
|
||||
"""Modality lighting values DTO."""
|
||||
|
||||
high: float
|
||||
medium: float
|
||||
low: float
|
||||
@@ -0,0 +1,10 @@
|
||||
"""Definition of PointOfViewValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class PointOfViewValues(ValuesModel):
|
||||
"""Point-of-view values DTO."""
|
||||
|
||||
frontal: float
|
||||
oblique: float
|
||||
@@ -0,0 +1,13 @@
|
||||
"""Definition of SalienceValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class SalienceValues(ValuesModel):
|
||||
"""Salience values DTO."""
|
||||
|
||||
size: float
|
||||
colour: float
|
||||
tone: float
|
||||
form: float
|
||||
positioning: float
|
||||
@@ -0,0 +1,60 @@
|
||||
"""Definition of ValuesModel base class."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import random
|
||||
|
||||
from pydantic import BaseModel, ConfigDict
|
||||
from torch import Tensor
|
||||
|
||||
|
||||
class ValuesModel(BaseModel):
|
||||
"""ValuesModel base class."""
|
||||
|
||||
model_config = ConfigDict(
|
||||
validate_assignment=True, # argument type checking
|
||||
frozen=True, # ensure data immutability
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def list_fields(cls) -> list[str]:
|
||||
"""List options that are stored as attributes."""
|
||||
return list(cls.model_fields.keys())
|
||||
|
||||
@classmethod
|
||||
def from_random(cls):
|
||||
"""Instantiate with random numbers."""
|
||||
kwargs = {field: random.random() for field in cls.list_fields()}
|
||||
return cls(**kwargs)
|
||||
|
||||
@classmethod
|
||||
def from_choice(cls, option: str) -> ValuesModel:
|
||||
"""Instantiate from choice."""
|
||||
assert isinstance(option, str)
|
||||
assert len(option) > 0
|
||||
allowed_options_list = cls.list_fields()
|
||||
if option not in allowed_options_list:
|
||||
raise ValueError(f'option {option} must be in {allowed_options_list}')
|
||||
# generate field values
|
||||
kwargs = {field: 0 for field in allowed_options_list}
|
||||
# set chosen value to max probability
|
||||
kwargs[option] = 1
|
||||
return cls(**kwargs)
|
||||
|
||||
@classmethod
|
||||
def from_tensor(cls, tensor: Tensor):
|
||||
"""Instantiate from list of values."""
|
||||
assert tensor.size(dim=0) == 1, f'tensor batch larger than 1: {tensor}'
|
||||
data_list = [float(t.item()) for t in tensor[0]]
|
||||
kwargs = dict(zip(cls.list_fields(), data_list))
|
||||
return cls(**kwargs)
|
||||
|
||||
def highest_score_field(self) -> str:
|
||||
"""Return name of field with highest score."""
|
||||
model_dict = self.model_dump()
|
||||
return max(model_dict, key=lambda k: model_dict[k])
|
||||
|
||||
def highest_score_value(self) -> float:
|
||||
"""Return value of field with highest score."""
|
||||
model_dict = self.model_dump()
|
||||
return max(model_dict.values())
|
||||
@@ -0,0 +1,14 @@
|
||||
"""Definition of VisualCommunicationData DTO."""
|
||||
|
||||
from pydantic import Field
|
||||
|
||||
from ..type_checking_base_model import TypeCheckingBaseModel
|
||||
from .visual_communication_values import VisualCommunicationValues
|
||||
|
||||
|
||||
class VisualCommunicationData(TypeCheckingBaseModel):
|
||||
"""Visual communication data class."""
|
||||
|
||||
name: str = Field(min_length=1)
|
||||
annotation: VisualCommunicationValues | None = None
|
||||
prediction: VisualCommunicationValues | None = None
|
||||
@@ -0,0 +1,49 @@
|
||||
"""Definition of VisualCommunicationValues DTO."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ..type_checking_base_model import TypeCheckingBaseModel
|
||||
from .angle_values import AngleValues
|
||||
from .contact_values import ContactValues
|
||||
from .distance_values import DistanceValues
|
||||
from .framing_values import FramingValues
|
||||
from .information_value_values import InformationValueValues
|
||||
from .modality_color_values import ModalityColorValues
|
||||
from .modality_depth_values import ModalityDepthValues
|
||||
from .modality_lighting_values import ModalityLightingValues
|
||||
from .point_of_view_values import PointOfViewValues
|
||||
from .salience_values import SalienceValues
|
||||
from .visual_syntax_values import VisualSyntaxValues
|
||||
|
||||
|
||||
class VisualCommunicationValues(TypeCheckingBaseModel):
|
||||
"""Visual communication values class."""
|
||||
|
||||
visual_syntax: VisualSyntaxValues
|
||||
contact: ContactValues
|
||||
angle: AngleValues
|
||||
point_of_view: PointOfViewValues
|
||||
distance: DistanceValues
|
||||
modality_lighting: ModalityLightingValues
|
||||
modality_color: ModalityColorValues
|
||||
modality_depth: ModalityDepthValues
|
||||
information_value: InformationValueValues
|
||||
framing: FramingValues
|
||||
salience: SalienceValues
|
||||
|
||||
@classmethod
|
||||
def from_random(cls) -> VisualCommunicationValues:
|
||||
"""Create a random instance."""
|
||||
return cls(
|
||||
visual_syntax=VisualSyntaxValues.from_random(),
|
||||
contact=ContactValues.from_random(),
|
||||
angle=AngleValues.from_random(),
|
||||
point_of_view=PointOfViewValues.from_random(),
|
||||
distance=DistanceValues.from_random(),
|
||||
modality_lighting=ModalityLightingValues.from_random(),
|
||||
modality_color=ModalityColorValues.from_random(),
|
||||
modality_depth=ModalityDepthValues.from_random(),
|
||||
information_value=InformationValueValues.from_random(),
|
||||
framing=FramingValues.from_random(),
|
||||
salience=SalienceValues.from_random(),
|
||||
)
|
||||
@@ -0,0 +1,26 @@
|
||||
"""Definition of VisualSyntaxValues DTO."""
|
||||
|
||||
from .values_model import ValuesModel
|
||||
|
||||
|
||||
class VisualSyntaxValues(ValuesModel):
|
||||
"""Visual syntax values DTO."""
|
||||
|
||||
non_transactional_action: float
|
||||
non_transactional_reaction: float
|
||||
unidirectional_transactional_action: float
|
||||
unidirectional_transactional_reaction: float
|
||||
bidirectional_transactional_action: float
|
||||
bidirectional_transactional_reaction: float
|
||||
conversion: float
|
||||
speech_process: float
|
||||
classification_overt_taxonomy: float
|
||||
analytical_exhaustive: float
|
||||
analytical_disarranged: float
|
||||
analytical_temporal: float
|
||||
analytical_distributed: float
|
||||
analytical_topological: float
|
||||
analytical_exploded: float
|
||||
analytical_inclusive: float
|
||||
symbolic_suggestive: float
|
||||
symbolic_attributive: float
|
||||
@@ -0,0 +1,74 @@
|
||||
"""Definition of ImageRepository class."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
|
||||
from PIL import Image
|
||||
|
||||
from .dto import ImageData
|
||||
from .implementations import MinioImplementation
|
||||
from .interfaces import ImageInterface
|
||||
|
||||
|
||||
class ImageRepository(ImageInterface, MinioImplementation):
|
||||
"""Image repository class that handles CRUD functionality for
|
||||
VisualData."""
|
||||
|
||||
def __enter__(self) -> ImageRepository:
|
||||
self.connect()
|
||||
return self
|
||||
|
||||
@staticmethod
|
||||
def _build_path(image_name: str) -> Path:
|
||||
"""Build object path."""
|
||||
assert isinstance(image_name, str)
|
||||
path = Path('images') / image_name
|
||||
return path
|
||||
|
||||
def get_data(self, image_name: str) -> ImageData | None:
|
||||
"""Get Visual data."""
|
||||
assert isinstance(image_name, str)
|
||||
assert len(image_name) > 0
|
||||
# build path
|
||||
path = self._build_path(image_name)
|
||||
# get object from bucket
|
||||
buffer = self._get(path)
|
||||
# handle if no data found
|
||||
if not buffer:
|
||||
return None
|
||||
# convert data
|
||||
image = Image.open(buffer)
|
||||
data = ImageData(image=image, name=image_name)
|
||||
return data
|
||||
|
||||
def put_data(self, data: ImageData) -> None:
|
||||
"""Put visual data."""
|
||||
assert isinstance(data, ImageData)
|
||||
# build path
|
||||
path = self._build_path(data.name)
|
||||
# save image to buffer
|
||||
buffer = BytesIO()
|
||||
data.image.save(buffer, 'png')
|
||||
# put object in bucket
|
||||
self._put(path, buffer)
|
||||
|
||||
def remove_data(self, image_name: str) -> None:
|
||||
"""Remove visual data."""
|
||||
assert isinstance(image_name, str)
|
||||
assert len(image_name) > 0
|
||||
# build path
|
||||
path = self._build_path(image_name)
|
||||
# remove object
|
||||
self._delete(path)
|
||||
|
||||
def list_names(self) -> list[str]:
|
||||
"""List names of all images."""
|
||||
# build path
|
||||
path = self._build_path('')
|
||||
# list object paths
|
||||
obj_path_list = self._list_objects(path)
|
||||
# strip prefix
|
||||
name_list = [obj_path.split('/')[-1] for obj_path in obj_path_list]
|
||||
return name_list
|
||||
@@ -0,0 +1,2 @@
|
||||
from .minio_implementation import MinioImplementation
|
||||
from .mongo_implementation import MongoImplementation
|
||||
@@ -0,0 +1,186 @@
|
||||
"""MinIO implementation."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from traceback import format_exc
|
||||
|
||||
from minio import Minio
|
||||
|
||||
from shared.utils import check_env
|
||||
|
||||
from ..interfaces import DatabaseInterface
|
||||
|
||||
|
||||
class MinioImplementation(DatabaseInterface):
|
||||
"""MinIO basic CRUD implementation."""
|
||||
|
||||
def __init__(self):
|
||||
# ensure necessary env vars available
|
||||
var_list = {
|
||||
'MINIO_ENDPOINT',
|
||||
'MINIO_ACCESS_KEY',
|
||||
'MINIO_SECRET_KEY',
|
||||
'MINIO_BUCKET_NAME',
|
||||
}
|
||||
check_env(var_list)
|
||||
# prepare internal variables
|
||||
self._client: Minio | None = None
|
||||
self._bucket_name: str | None = None
|
||||
|
||||
def connect(self):
|
||||
"""Connect to MinIO server."""
|
||||
# prepare arguments
|
||||
minio_endpoint = str(os.getenv('MINIO_ENDPOINT'))
|
||||
minio_access_key = str(os.getenv('MINIO_ACCESS_KEY'))
|
||||
minio_secret_key = str(os.getenv('MINIO_SECRET_KEY'))
|
||||
minio_bucket_name = str(os.getenv('MINIO_BUCKET_NAME'))
|
||||
# connect client
|
||||
client = Minio(
|
||||
endpoint=minio_endpoint,
|
||||
access_key=minio_access_key,
|
||||
secret_key=minio_secret_key,
|
||||
secure=False,
|
||||
)
|
||||
# ensure bucket exists
|
||||
if not client.bucket_exists(bucket_name=minio_bucket_name):
|
||||
logging.debug('creating bucket: %s', minio_bucket_name)
|
||||
client.make_bucket(bucket_name=minio_bucket_name)
|
||||
# persist state
|
||||
self._client = client
|
||||
self._bucket_name = minio_bucket_name
|
||||
|
||||
def close(self) -> None:
|
||||
"""Close connection to MinIO server.
|
||||
|
||||
N.B. MinIO connection cannot be closed manually.
|
||||
"""
|
||||
self._client = None
|
||||
self._bucket_name = None
|
||||
|
||||
def connected(self):
|
||||
"""Check connection to Minio."""
|
||||
res = isinstance(self._client, Minio)
|
||||
logging.debug(res)
|
||||
return res
|
||||
|
||||
def __enter__(self) -> MinioImplementation:
|
||||
self.connect()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_val, exc_tb) -> None:
|
||||
if any(
|
||||
(
|
||||
exc_type is not None,
|
||||
exc_val is not None,
|
||||
exc_tb is not None,
|
||||
),
|
||||
):
|
||||
logging.error('error while exiting context')
|
||||
self.close()
|
||||
|
||||
def _put(
|
||||
self,
|
||||
path: Path,
|
||||
buffer: BytesIO,
|
||||
) -> None:
|
||||
"""Save in-memory buffer as object in MinIO."""
|
||||
assert isinstance(path, Path)
|
||||
assert isinstance(buffer, BytesIO)
|
||||
assert isinstance(self._client, Minio)
|
||||
assert isinstance(self._bucket_name, str)
|
||||
# prepare for saving
|
||||
num_bytes = len(buffer.getvalue())
|
||||
buffer.seek(0)
|
||||
# send data to bucket
|
||||
try:
|
||||
self._client.put_object(
|
||||
bucket_name=self._bucket_name,
|
||||
object_name=path.as_posix(),
|
||||
length=num_bytes,
|
||||
data=buffer,
|
||||
)
|
||||
logging.debug('saved data to %s', path)
|
||||
except Exception as exc:
|
||||
logging.error('failed saving data to MinIO')
|
||||
raise exc
|
||||
|
||||
def _get(
|
||||
self,
|
||||
path: Path,
|
||||
) -> BytesIO | None:
|
||||
"""Get object from MinIO as in-memory buffer."""
|
||||
assert isinstance(path, Path)
|
||||
assert isinstance(self._client, Minio)
|
||||
assert isinstance(self._bucket_name, str)
|
||||
try:
|
||||
# make request
|
||||
response = self._client.get_object(
|
||||
bucket_name=self._bucket_name,
|
||||
object_name=path.as_posix(),
|
||||
)
|
||||
assert response.status == 200
|
||||
# get buffer
|
||||
buffer = BytesIO()
|
||||
chunk_size = 2**14
|
||||
while chunk := response.read(chunk_size):
|
||||
buffer.write(chunk)
|
||||
buffer.seek(0)
|
||||
logging.debug('got %s', path)
|
||||
return buffer
|
||||
except Exception:
|
||||
logging.error('failed getting data from MinIO')
|
||||
logging.debug(format_exc())
|
||||
return None
|
||||
finally:
|
||||
# close connection if established
|
||||
if 'response' in locals():
|
||||
response.close()
|
||||
response.release_conn()
|
||||
|
||||
def _delete(
|
||||
self,
|
||||
path: Path,
|
||||
) -> None:
|
||||
"""Delete object from MinIO."""
|
||||
assert isinstance(path, Path)
|
||||
assert isinstance(self._client, Minio)
|
||||
assert isinstance(self._bucket_name, str)
|
||||
# remove object
|
||||
try:
|
||||
self._client.remove_object(
|
||||
bucket_name=self._bucket_name,
|
||||
object_name=path.as_posix(),
|
||||
)
|
||||
logging.debug('deleted %s', path)
|
||||
except Exception as exc:
|
||||
logging.error('failed deleting %s', path)
|
||||
logging.debug(format_exc())
|
||||
raise exc
|
||||
|
||||
def _list_objects(
|
||||
self,
|
||||
path: Path,
|
||||
) -> list[str]:
|
||||
"""List objects in bucket under path."""
|
||||
assert isinstance(path, Path)
|
||||
assert isinstance(self._client, Minio)
|
||||
assert isinstance(self._bucket_name, str)
|
||||
try:
|
||||
# list objects
|
||||
obj_list = self._client.list_objects(
|
||||
bucket_name=self._bucket_name,
|
||||
prefix=path.as_posix(),
|
||||
recursive=True,
|
||||
)
|
||||
# extract info
|
||||
name_list = [obj.object_name for obj in obj_list]
|
||||
logging.debug('got %s objects matching %s', len(name_list), path)
|
||||
return name_list
|
||||
except Exception as exc:
|
||||
logging.error('failed listing objects under %s', path)
|
||||
logging.debug(format_exc())
|
||||
raise exc
|
||||
@@ -0,0 +1,145 @@
|
||||
"""Mongo implementation."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import traceback
|
||||
|
||||
from pymongo import MongoClient
|
||||
from pymongo.collection import Collection
|
||||
from pymongo.database import Database
|
||||
from pymongo.errors import ServerSelectionTimeoutError
|
||||
|
||||
from shared.utils import check_env
|
||||
|
||||
from ..interfaces import DatabaseInterface
|
||||
|
||||
|
||||
class MongoImplementation(DatabaseInterface):
|
||||
"""MongoDB basic CRUD implementation."""
|
||||
|
||||
def __init__(self):
|
||||
# ensure necessary env vars available
|
||||
var_list = {
|
||||
'MONGO_ENDPOINT',
|
||||
'MONGO_DB',
|
||||
'MONGO_COLLECTION',
|
||||
}
|
||||
check_env(var_list)
|
||||
# prepare internal variables
|
||||
self.client: MongoClient | None = None
|
||||
self.db: Database | None = None
|
||||
self.collection: Collection | None = None
|
||||
|
||||
def connect(self) -> None:
|
||||
"""Connect to Mongo server."""
|
||||
# prepare arguments
|
||||
mongo_endpoint = str(os.getenv('MONGO_ENDPOINT'))
|
||||
mongo_database = str(os.getenv('MONGO_DB'))
|
||||
mongo_collection = str(os.getenv('MONGO_COLLECTION'))
|
||||
# connect client
|
||||
client: MongoClient = MongoClient(mongo_endpoint)
|
||||
database = client[mongo_database]
|
||||
collection = database[mongo_collection]
|
||||
# set unique index on 'name'
|
||||
collection.create_index(keys='name', unique=True)
|
||||
# persist state
|
||||
self._client = client
|
||||
self._database = database
|
||||
self._collection = collection
|
||||
|
||||
def close(self) -> None:
|
||||
"""Close connection to Mongo server."""
|
||||
self._client.close()
|
||||
self._client = None # type: ignore
|
||||
self._database = None # type: ignore
|
||||
self._collection = None # type: ignore
|
||||
|
||||
def connected(self) -> bool:
|
||||
"""Check connection to Mongo."""
|
||||
if self._client is None:
|
||||
return False
|
||||
try:
|
||||
# trigger fetch data
|
||||
_ = self._client.server_info()
|
||||
res = True
|
||||
except ServerSelectionTimeoutError:
|
||||
res = False
|
||||
logging.debug(res)
|
||||
return res
|
||||
|
||||
def __enter__(self) -> MongoImplementation:
|
||||
self.connect()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_val, exc_tb) -> None:
|
||||
if any(
|
||||
(
|
||||
exc_type is not None,
|
||||
exc_val is not None,
|
||||
exc_tb is not None,
|
||||
),
|
||||
):
|
||||
logging.error('error while exiting context')
|
||||
traceback.print_exception(exc_type, exc_val, exc_tb)
|
||||
self.close()
|
||||
|
||||
def _save(
|
||||
self,
|
||||
data: dict,
|
||||
query: dict,
|
||||
) -> None:
|
||||
"""Save document in Mongo."""
|
||||
assert isinstance(data, dict)
|
||||
assert isinstance(query, dict)
|
||||
assert 'name' in data
|
||||
assert self.connected()
|
||||
self._collection.update_one(
|
||||
filter=query,
|
||||
update={
|
||||
'$set': data.copy(),
|
||||
},
|
||||
upsert=True,
|
||||
)
|
||||
logging.debug('Save %s', data)
|
||||
|
||||
def _get(
|
||||
self,
|
||||
query: dict,
|
||||
) -> dict | None:
|
||||
"""Get document from Mongo."""
|
||||
assert isinstance(query, dict)
|
||||
assert self.connected()
|
||||
doc = self._collection.find_one(query, projection={'_id': False})
|
||||
logging.debug('Found %s', doc)
|
||||
return doc
|
||||
|
||||
def _delete(
|
||||
self,
|
||||
query: dict,
|
||||
) -> None:
|
||||
"""Remove document from Mongo."""
|
||||
assert isinstance(query, dict)
|
||||
assert self.connected()
|
||||
doc = self._collection.delete_one(query)
|
||||
logging.debug('Deleted %s', doc)
|
||||
|
||||
def _list_documents(self, key='name') -> list[str]:
|
||||
"""List documents in Mongo."""
|
||||
assert isinstance(key, str)
|
||||
assert len(key) > 0
|
||||
# build query
|
||||
doc_list = list(
|
||||
self._collection.find(
|
||||
filter={},
|
||||
projection={
|
||||
'_id': False,
|
||||
key: True,
|
||||
},
|
||||
),
|
||||
)
|
||||
# extract values
|
||||
value_list = [doc[key] for doc in doc_list]
|
||||
logging.debug('Got %s document(s)', len(doc_list))
|
||||
return value_list
|
||||
@@ -0,0 +1,4 @@
|
||||
from .database_interface import DatabaseInterface
|
||||
from .image_interface import ImageInterface
|
||||
from .model_interface import ModelInterface
|
||||
from .visual_communication_interface import VisualCommunicationInterface
|
||||
@@ -0,0 +1,28 @@
|
||||
"""Definition of DatabaseInterface class."""
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
|
||||
|
||||
class DatabaseInterface(ABC):
|
||||
"""Interface base class adding 'connect', 'close' and context
|
||||
functionalities."""
|
||||
|
||||
@abstractmethod
|
||||
def connect(self):
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def close(self):
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def connected(self) -> bool:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def __enter__(self):
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def __exit__(self, exc_type, exc_val, exc_tb):
|
||||
raise NotImplementedError()
|
||||
@@ -0,0 +1,26 @@
|
||||
"""Definition of ImageInterface."""
|
||||
|
||||
from abc import abstractmethod
|
||||
|
||||
from ..dto import ImageData
|
||||
from .database_interface import DatabaseInterface
|
||||
|
||||
|
||||
class ImageInterface(DatabaseInterface):
|
||||
"""Image interface class."""
|
||||
|
||||
@abstractmethod
|
||||
def get_data(self, image_name: str) -> ImageData | None:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def put_data(self, data: ImageData) -> None:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def remove_data(self, image_name: str) -> None:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def list_names(self) -> list[str]:
|
||||
raise NotImplementedError()
|
||||
@@ -0,0 +1,26 @@
|
||||
"""Definition of ModelInterface."""
|
||||
|
||||
from abc import abstractmethod
|
||||
|
||||
from ..dto import ModelData
|
||||
from .database_interface import DatabaseInterface
|
||||
|
||||
|
||||
class ModelInterface(DatabaseInterface):
|
||||
"""Model interface class."""
|
||||
|
||||
@abstractmethod
|
||||
def get_data(self, object_name: str) -> ModelData | None:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def put_data(self, data: ModelData) -> None:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def remove_data(self, object_name: str) -> None:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def list_names(self) -> list[str]:
|
||||
raise NotImplementedError()
|
||||
@@ -0,0 +1,26 @@
|
||||
"""Definition of VisualCommunicationInterface."""
|
||||
|
||||
from abc import abstractmethod
|
||||
|
||||
from ..dto import VisualCommunicationData
|
||||
from .database_interface import DatabaseInterface
|
||||
|
||||
|
||||
class VisualCommunicationInterface(DatabaseInterface):
|
||||
"""Visual communication interface class."""
|
||||
|
||||
@abstractmethod
|
||||
def get_data(self, name: str) -> VisualCommunicationData | None:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def put_data(self, data: VisualCommunicationData) -> None:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def remove_data(self, name: str) -> None:
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def list_names(self) -> list[str]:
|
||||
raise NotImplementedError()
|
||||
@@ -0,0 +1,77 @@
|
||||
"""Definition of ModelRepository class."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from .dto import HexadecimalString, ModelData
|
||||
from .implementations import MinioImplementation
|
||||
from .interfaces import ModelInterface
|
||||
|
||||
|
||||
class ModelRepository(ModelInterface, MinioImplementation):
|
||||
"""Model repository class that handles CRUD functionality for ModelData."""
|
||||
|
||||
def __enter__(self) -> ModelRepository:
|
||||
self.connect()
|
||||
return self
|
||||
|
||||
@staticmethod
|
||||
def _prefix() -> Path:
|
||||
"""Object name prefix."""
|
||||
return Path('models')
|
||||
|
||||
@classmethod
|
||||
def _build_object_name(cls, data: ModelData) -> str:
|
||||
"""Build object name from data."""
|
||||
return f'{data.class_name}-{data.buffer_checksum}'
|
||||
|
||||
def get_data(self, object_name: str) -> ModelData | None:
|
||||
"""Get model data."""
|
||||
assert isinstance(object_name, str)
|
||||
# build path
|
||||
path = self._prefix() / object_name
|
||||
# get object from bucket
|
||||
buffer = self._get(path)
|
||||
# handle if no data found
|
||||
if not buffer:
|
||||
return None
|
||||
# extract info
|
||||
class_name, buffer_checksum_str = object_name.split('-')
|
||||
# convert data
|
||||
buffer_checksum = HexadecimalString(buffer_checksum_str)
|
||||
# instantiate data
|
||||
data = ModelData(
|
||||
buffer=buffer,
|
||||
buffer_checksum=buffer_checksum,
|
||||
class_name=class_name,
|
||||
)
|
||||
return data
|
||||
|
||||
def put_data(self, data: ModelData) -> None:
|
||||
"""Put model data."""
|
||||
assert isinstance(data, ModelData)
|
||||
# build object name
|
||||
object_name = self._build_object_name(data)
|
||||
# build path
|
||||
path = self._prefix() / object_name
|
||||
# put object in bucket
|
||||
self._put(path, data.buffer)
|
||||
|
||||
def remove_data(self, object_name: str) -> None:
|
||||
"""Remove model data."""
|
||||
assert isinstance(object_name, str)
|
||||
# build path
|
||||
path = self._prefix() / object_name
|
||||
# remove object
|
||||
self._delete(path)
|
||||
|
||||
def list_names(self) -> list[str]:
|
||||
"""List names of all models."""
|
||||
# build path
|
||||
path = self._prefix()
|
||||
# list object paths
|
||||
obj_path_list = self._list_objects(path)
|
||||
# strip prefix
|
||||
name_list = [obj_path.split('/')[-1] for obj_path in obj_path_list]
|
||||
return name_list
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Definition of VisualCommunicationRepository class."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .dto import VisualCommunicationData
|
||||
from .implementations import MongoImplementation
|
||||
from .interfaces import VisualCommunicationInterface
|
||||
|
||||
|
||||
class VisualCommunicationRepository(VisualCommunicationInterface, MongoImplementation):
|
||||
"""Visual communication repository class that handles CRUD functionality
|
||||
for VisualCommunicationData."""
|
||||
|
||||
def __enter__(self) -> VisualCommunicationRepository:
|
||||
self.connect()
|
||||
return self
|
||||
|
||||
def get_data(self, name: str) -> VisualCommunicationData | None:
|
||||
"""Get visual communication data."""
|
||||
assert isinstance(name, str)
|
||||
assert len(name) > 0
|
||||
# build query
|
||||
query = {'name': name}
|
||||
# get document from mongo
|
||||
doc = self._get(query)
|
||||
# handle if no data found
|
||||
if doc is None:
|
||||
return None
|
||||
# instantiate object
|
||||
data = VisualCommunicationData(**doc)
|
||||
return data
|
||||
|
||||
def put_data(self, data: VisualCommunicationData) -> None:
|
||||
"""Put visual communication data."""
|
||||
assert isinstance(data, VisualCommunicationData)
|
||||
# convert to dict
|
||||
data_dict: dict = data.model_dump(mode='json')
|
||||
# build query
|
||||
query = {'name': data.name}
|
||||
# save to mongo
|
||||
self._save(data_dict, query)
|
||||
|
||||
def remove_data(self, name: str) -> None:
|
||||
"""Remove visual communication data."""
|
||||
assert isinstance(name, str)
|
||||
assert len(name) > 0
|
||||
# build query
|
||||
query = {'name': name}
|
||||
# remove document from mongo
|
||||
self._delete(query)
|
||||
|
||||
def list_names(self) -> list[str]:
|
||||
"""List names of all documents."""
|
||||
# list documents
|
||||
name_list = self._list_documents(key='name')
|
||||
return name_list
|
||||
@@ -0,0 +1,403 @@
|
||||
"""Integration tests configuration."""
|
||||
|
||||
import os
|
||||
import random
|
||||
from collections.abc import Iterator
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
|
||||
import minio
|
||||
import pytest
|
||||
from dotenv import load_dotenv
|
||||
from PIL import Image
|
||||
from pymongo import MongoClient
|
||||
|
||||
from model.src.models import VisualCommunicationModel
|
||||
from shared.repositories import (
|
||||
HexadecimalString,
|
||||
ImageData,
|
||||
ImageRepository,
|
||||
ModelData,
|
||||
ModelRepository,
|
||||
VisualCommunicationData,
|
||||
VisualCommunicationRepository,
|
||||
VisualCommunicationValues,
|
||||
)
|
||||
from shared.repositories.src.implementations import (
|
||||
MinioImplementation,
|
||||
MongoImplementation,
|
||||
)
|
||||
|
||||
# set random seed for reproducibility
|
||||
random.seed(13)
|
||||
|
||||
# define test environment variables
|
||||
necessary_env_vars = {
|
||||
'MINIO_ENDPOINT',
|
||||
'MINIO_ACCESS_KEY',
|
||||
'MINIO_SECRET_KEY',
|
||||
'MONGO_ENDPOINT',
|
||||
}
|
||||
env_var_map = {
|
||||
'MINIO_BUCKET_NAME': 'test-bucket',
|
||||
'MINIO_OBJECT_NAME': 'test-object',
|
||||
'MINIO_IMAGE_NAME': 'test-image',
|
||||
'MONGO_DB': 'test-db',
|
||||
'MONGO_COLLECTION': 'test-collection',
|
||||
}
|
||||
|
||||
|
||||
@pytest.fixture(scope='session', autouse=True)
|
||||
def setup_env(
|
||||
request: pytest.FixtureRequest,
|
||||
) -> None:
|
||||
"""Populate environment with variables used for testing."""
|
||||
# load in optional local test environment variables
|
||||
test_env_path = Path(__file__).parent.parent.parent.parent.parent / 'test.env'
|
||||
load_dotenv(test_env_path)
|
||||
# check if necessary env vars are set
|
||||
for key in necessary_env_vars:
|
||||
assert key in os.environ, f'{key} not set'
|
||||
# set env vars unique to this test
|
||||
for key, val in env_var_map.items():
|
||||
# set env var
|
||||
os.environ[key] = val
|
||||
|
||||
# ensure cleanup
|
||||
def cleanup_env():
|
||||
for key in env_var_map:
|
||||
_ = os.environ.pop(key, default=None)
|
||||
|
||||
request.addfinalizer(cleanup_env)
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
def raw_minio_client(
|
||||
setup_env,
|
||||
) -> Iterator[minio.Minio]:
|
||||
"""Raw Minio client fixture."""
|
||||
# prepare arguments
|
||||
minio_endpoint = str(os.getenv('MINIO_ENDPOINT'))
|
||||
minio_access_key = str(os.getenv('MINIO_ACCESS_KEY'))
|
||||
minio_secret_key = str(os.getenv('MINIO_SECRET_KEY'))
|
||||
minio_bucket_name = str(os.getenv('MINIO_BUCKET_NAME'))
|
||||
# connect client
|
||||
client = minio.Minio(
|
||||
endpoint=minio_endpoint,
|
||||
access_key=minio_access_key,
|
||||
secret_key=minio_secret_key,
|
||||
secure=False,
|
||||
)
|
||||
# ensure bucket exists
|
||||
if not client.bucket_exists(bucket_name=minio_bucket_name):
|
||||
client.make_bucket(bucket_name=minio_bucket_name)
|
||||
# expose client
|
||||
yield client
|
||||
# cleanup
|
||||
object_list = client.list_objects(minio_bucket_name, recursive=True)
|
||||
for obj in object_list:
|
||||
client.remove_object(
|
||||
bucket_name=obj.bucket_name,
|
||||
object_name=obj.object_name,
|
||||
)
|
||||
client.remove_bucket(minio_bucket_name)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def minio_client(
|
||||
setup_env,
|
||||
) -> Iterator[MinioImplementation]:
|
||||
"""MinioImplementation fixture."""
|
||||
# instantiate and connect client
|
||||
minio_client = MinioImplementation()
|
||||
minio_client.connect()
|
||||
# expose client
|
||||
yield minio_client
|
||||
# cleanup
|
||||
object_name_list = minio_client._list_objects(Path('*'))
|
||||
for name in object_name_list:
|
||||
minio_client._delete(Path(name))
|
||||
minio_client.close()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def buffer() -> Iterator[BytesIO]:
|
||||
"""Bytes buffer fixture."""
|
||||
# generate reproducible random data
|
||||
data = random.randbytes(n=2**21) # 2 MB
|
||||
# convert data
|
||||
buffer = BytesIO(data)
|
||||
# expose buffer
|
||||
yield buffer
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def buffer_in_minio(
|
||||
raw_minio_client: minio.Minio,
|
||||
buffer: BytesIO,
|
||||
) -> Iterator[tuple[Path, BytesIO]]:
|
||||
"""Buffer in Minio fixture."""
|
||||
# prepare arguments
|
||||
minio_bucket_name = str(os.getenv('MINIO_BUCKET_NAME'))
|
||||
minio_object_name = str(os.getenv('MINIO_OBJECT_NAME'))
|
||||
# prepare for saving
|
||||
num_bytes = len(buffer.getvalue())
|
||||
buffer.seek(0)
|
||||
# put data in bucket
|
||||
raw_minio_client.put_object(
|
||||
bucket_name=minio_bucket_name,
|
||||
object_name=minio_object_name,
|
||||
length=num_bytes,
|
||||
data=buffer,
|
||||
)
|
||||
# expose data
|
||||
yield Path(minio_object_name), buffer
|
||||
# cleanup
|
||||
raw_minio_client.remove_object(
|
||||
bucket_name=minio_bucket_name,
|
||||
object_name=minio_object_name,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def image_data() -> Iterator[ImageData]:
|
||||
"""Image data fixture."""
|
||||
# prepare arguments
|
||||
name = str(os.getenv('MINIO_IMAGE_NAME'))
|
||||
image = Image.new(mode='RGB', size=(480, 480))
|
||||
# instantiate data
|
||||
image_data = ImageData(image=image, name=name)
|
||||
# expose data
|
||||
yield image_data
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def image_data_in_minio(
|
||||
raw_minio_client: minio.Minio,
|
||||
image_data: ImageData,
|
||||
) -> Iterator[ImageData]:
|
||||
"""Image data in Minio fixture."""
|
||||
# prepare arguments
|
||||
minio_bucket_name = str(os.getenv('MINIO_BUCKET_NAME'))
|
||||
image_name = image_data.name
|
||||
# build object path
|
||||
object_path = ImageRepository._build_path(image_name)
|
||||
# save image to buffer
|
||||
buffer = BytesIO()
|
||||
image_data.image.save(buffer, 'png')
|
||||
# prepare for saving
|
||||
num_bytes = len(buffer.getvalue())
|
||||
buffer.seek(0)
|
||||
# put data in bucket
|
||||
raw_minio_client.put_object(
|
||||
bucket_name=minio_bucket_name,
|
||||
object_name=object_path.as_posix(),
|
||||
length=num_bytes,
|
||||
data=buffer,
|
||||
)
|
||||
# expose data
|
||||
yield image_data
|
||||
# cleanup
|
||||
raw_minio_client.remove_object(
|
||||
bucket_name=minio_bucket_name,
|
||||
object_name=object_path.as_posix(),
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
def image_repo(
|
||||
setup_env,
|
||||
) -> Iterator[ImageRepository]:
|
||||
"""Image repository fixture."""
|
||||
# define test environment variables
|
||||
assert 'MINIO_ENDPOINT' in os.environ, 'MINIO_ENDPOINT not set'
|
||||
assert 'MINIO_ACCESS_KEY' in os.environ, 'MINIO_ACCESS_KEY not set'
|
||||
assert 'MINIO_SECRET_KEY' in os.environ, 'MINIO_SECRET_KEY not set'
|
||||
assert 'MONGO_ENDPOINT' in os.environ, 'MONGO_ENDPOINT not set'
|
||||
repo = ImageRepository()
|
||||
repo.connect()
|
||||
yield repo
|
||||
repo.close()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def model_data() -> Iterator[ModelData]:
|
||||
"""Model data fixture."""
|
||||
# prepare arguments
|
||||
vis_com_model = VisualCommunicationModel().to('cpu')
|
||||
class_name = type(vis_com_model).__name__
|
||||
buffer = ModelData.model_to_buffer(vis_com_model)
|
||||
buffer_checksum = HexadecimalString('77dcab1769563654a6e24f92d40f29bd')
|
||||
# instantiate data
|
||||
model_data = ModelData(
|
||||
buffer=buffer,
|
||||
buffer_checksum=buffer_checksum,
|
||||
class_name=class_name,
|
||||
)
|
||||
# expose model
|
||||
yield model_data
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def model_data_in_minio(
|
||||
raw_minio_client: minio.Minio,
|
||||
model_data: ModelData,
|
||||
) -> Iterator[ModelData]:
|
||||
"""Model data in Minio fixture."""
|
||||
# prepare arguments
|
||||
minio_bucket_name = str(os.getenv('MINIO_BUCKET_NAME'))
|
||||
object_name = ModelRepository._build_object_name(model_data)
|
||||
buffer = model_data.buffer
|
||||
# build object path
|
||||
object_path = ModelRepository._prefix() / object_name
|
||||
# prepare for saving
|
||||
num_bytes = len(buffer.getvalue())
|
||||
buffer.seek(0)
|
||||
# put data in bucket
|
||||
raw_minio_client.put_object(
|
||||
bucket_name=minio_bucket_name,
|
||||
object_name=object_path.as_posix(),
|
||||
length=num_bytes,
|
||||
data=buffer,
|
||||
)
|
||||
# expose data
|
||||
yield model_data
|
||||
# cleanup
|
||||
raw_minio_client.remove_object(
|
||||
bucket_name=minio_bucket_name,
|
||||
object_name=object_path.as_posix(),
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
def model_repo(
|
||||
setup_env,
|
||||
) -> Iterator[ModelRepository]:
|
||||
"""Model repository fixture."""
|
||||
repo = ModelRepository()
|
||||
repo.connect()
|
||||
yield repo
|
||||
repo.close()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def raw_mongo_client(
|
||||
setup_env,
|
||||
) -> Iterator[MongoClient]:
|
||||
"""Raw mongo client fixture."""
|
||||
# prepare arguments
|
||||
mongo_endpoint = str(os.getenv('MONGO_ENDPOINT'))
|
||||
mongo_database = str(os.getenv('MONGO_DB'))
|
||||
# connect client
|
||||
client: MongoClient = MongoClient(mongo_endpoint)
|
||||
_ = client[mongo_database]
|
||||
# expose client
|
||||
yield client
|
||||
# cleanup
|
||||
client.drop_database(mongo_database)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mongo_client(
|
||||
setup_env,
|
||||
) -> Iterator[MongoImplementation]:
|
||||
"""MongoImplementation fixture."""
|
||||
# instantiate and connect client
|
||||
mongo_client = MongoImplementation()
|
||||
mongo_client.connect()
|
||||
# expose client
|
||||
yield mongo_client
|
||||
# cleanup
|
||||
mongo_client._collection.drop()
|
||||
mongo_client.close()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def dictionary() -> Iterator[dict]:
|
||||
"""Dictionary fixture."""
|
||||
# prepare data
|
||||
data = {
|
||||
'name': 'test-dictionary',
|
||||
'str_key': 'value',
|
||||
'int_key': 100,
|
||||
'float_key': 3.14,
|
||||
'list_key': [1, 2, 3],
|
||||
'dict_key': {
|
||||
'nested_key': 'nested_value',
|
||||
},
|
||||
}
|
||||
# expose data
|
||||
yield data
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def dictionary_in_mongo(
|
||||
raw_mongo_client: MongoClient,
|
||||
dictionary: dict,
|
||||
) -> Iterator[dict]:
|
||||
"""Dictionary in Mongo fixture."""
|
||||
# prepare arguments
|
||||
database = str(os.getenv('MONGO_DB'))
|
||||
collection = str(os.getenv('MONGO_COLLECTION'))
|
||||
# save data
|
||||
_ = raw_mongo_client[database][collection].insert_one(dictionary.copy())
|
||||
# expose data
|
||||
yield dictionary
|
||||
# cleanup
|
||||
raw_mongo_client[database][collection].delete_one(dictionary)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def visual_communication_values() -> Iterator[VisualCommunicationValues]:
|
||||
"""Visual communication values fixture."""
|
||||
# instantiate with random values
|
||||
visual_communication_values = VisualCommunicationValues.from_random()
|
||||
# expose values
|
||||
yield visual_communication_values
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def visual_communication_data(
|
||||
visual_communication_values: VisualCommunicationValues,
|
||||
) -> Iterator[VisualCommunicationData]:
|
||||
"""Visual communication data fixture."""
|
||||
# prepare arguments
|
||||
name = 'test-visual-communication'
|
||||
annotation = visual_communication_values
|
||||
# instantiate data
|
||||
visual_communication_data = VisualCommunicationData(
|
||||
name=name,
|
||||
annotation=annotation,
|
||||
)
|
||||
# expose data
|
||||
yield visual_communication_data
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def visual_communication_data_in_mongo(
|
||||
raw_mongo_client: MongoClient,
|
||||
visual_communication_data: VisualCommunicationData,
|
||||
) -> Iterator[VisualCommunicationData]:
|
||||
"""Visual communication data in Mongo fixture."""
|
||||
# prepare arguments
|
||||
database = str(os.getenv('MONGO_DB'))
|
||||
collection = str(os.getenv('MONGO_COLLECTION'))
|
||||
# convert data
|
||||
dictionary = visual_communication_data.model_dump(mode='dict')
|
||||
# save data
|
||||
_ = raw_mongo_client[database][collection].insert_one(dictionary.copy())
|
||||
# expose data
|
||||
yield visual_communication_data
|
||||
# cleanup
|
||||
raw_mongo_client[database][collection].delete_one(dictionary)
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
def visual_communication_repo(
|
||||
setup_env,
|
||||
) -> Iterator[VisualCommunicationRepository]:
|
||||
"""Visual communication repository fixture."""
|
||||
repo = VisualCommunicationRepository()
|
||||
repo.connect()
|
||||
yield repo
|
||||
repo.close()
|
||||
@@ -0,0 +1,156 @@
|
||||
"""Integration tests for ImageRepository class."""
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
from PIL import Image
|
||||
|
||||
from shared.repositories import ImageData, ImageRepository
|
||||
|
||||
|
||||
def same_image(
|
||||
img_a: Image.Image,
|
||||
img_b: Image.Image,
|
||||
) -> bool:
|
||||
"""Check if two images contain the same data."""
|
||||
assert isinstance(img_a, Image.Image)
|
||||
assert isinstance(img_b, Image.Image)
|
||||
# check if images have a comparable number of channels
|
||||
if img_a.getbands() != img_b.getbands():
|
||||
return False
|
||||
# calculate pixel difference between images
|
||||
img_a_arr = np.asarray(img_a)
|
||||
img_b_arr = np.asarray(img_b)
|
||||
diff = np.subtract(img_a_arr, img_b_arr)
|
||||
if np.sum(diff) != 0:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def same_image_data(
|
||||
data_a: ImageData,
|
||||
data_b: ImageData,
|
||||
) -> bool:
|
||||
"""Check if two ImageData-objects contain the same data."""
|
||||
assert isinstance(data_a, ImageData)
|
||||
assert isinstance(data_b, ImageData)
|
||||
# compare names
|
||||
if data_a.name != data_b.name:
|
||||
return False
|
||||
# compare images
|
||||
if not same_image(data_a.image, data_b.image):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def test_should_have_context_handler():
|
||||
"""Test that class has a working context handler implemented."""
|
||||
# ACT
|
||||
with ImageRepository() as repo:
|
||||
# ASSERT
|
||||
assert repo.connected()
|
||||
|
||||
|
||||
def test_should_get_image_data(
|
||||
image_repo: ImageRepository,
|
||||
image_data_in_minio: ImageData,
|
||||
):
|
||||
"""Test getting image data."""
|
||||
# ARRANGE
|
||||
image_name = image_data_in_minio.name
|
||||
# ACT
|
||||
received_image_data = image_repo.get_data(image_name)
|
||||
# ASSERT
|
||||
assert received_image_data is not None
|
||||
assert isinstance(received_image_data, ImageData)
|
||||
assert same_image_data(image_data_in_minio, received_image_data)
|
||||
|
||||
|
||||
def test_should_get_none_when_no_image_data(
|
||||
image_repo: ImageRepository,
|
||||
image_data: ImageData,
|
||||
):
|
||||
"""Test getting None when no data is available."""
|
||||
# ARRANGE
|
||||
image_name = image_data.name
|
||||
# ACT
|
||||
received_image_data = image_repo.get_data(image_name)
|
||||
# ASSERT
|
||||
assert received_image_data is None
|
||||
|
||||
|
||||
def test_should_delete_image_data(
|
||||
image_repo: ImageRepository,
|
||||
image_data_in_minio: ImageData,
|
||||
):
|
||||
"""Test deleting image data."""
|
||||
# ARRANGE
|
||||
image_name = image_data_in_minio.name
|
||||
# ACT
|
||||
image_repo.remove_data(image_name)
|
||||
received_image_data = image_repo.get_data(image_name)
|
||||
# ASSERT
|
||||
assert received_image_data is None
|
||||
|
||||
|
||||
def test_should_put_image_data(
|
||||
image_repo: ImageRepository,
|
||||
image_data: ImageData,
|
||||
):
|
||||
"""Test putting image data."""
|
||||
# ARRANGE
|
||||
image_name = image_data.name
|
||||
# ACT
|
||||
image_repo.put_data(image_data)
|
||||
received_image_data = image_repo.get_data(image_name)
|
||||
# ASSERT
|
||||
assert received_image_data is not None
|
||||
assert same_image_data(image_data, received_image_data)
|
||||
|
||||
|
||||
def test_should_update_image_data(
|
||||
image_repo: ImageRepository,
|
||||
image_data_in_minio: ImageData,
|
||||
):
|
||||
"""Test updating image data."""
|
||||
# ARRANGE
|
||||
updated_image_data = image_data_in_minio.model_copy(
|
||||
update={
|
||||
'image': Image.new(mode='RGB', size=(480, 480), color='white'),
|
||||
},
|
||||
)
|
||||
image_name = updated_image_data.name
|
||||
# ACT
|
||||
image_repo.put_data(updated_image_data)
|
||||
received_image_data = image_repo.get_data(image_name)
|
||||
# ASSERT
|
||||
assert not same_image_data(image_data_in_minio, updated_image_data)
|
||||
assert received_image_data is not None
|
||||
assert same_image_data(updated_image_data, received_image_data)
|
||||
|
||||
|
||||
def test_should_list_names(
|
||||
image_repo: ImageRepository,
|
||||
image_data_in_minio: ImageData,
|
||||
):
|
||||
"""Test get all image names."""
|
||||
# ARRANGE
|
||||
updated_image_data = image_data_in_minio.model_copy(
|
||||
update={
|
||||
'name': 'updated-test-image',
|
||||
},
|
||||
)
|
||||
image_repo.put_data(updated_image_data)
|
||||
expected_name_list = [
|
||||
image_data_in_minio.name,
|
||||
updated_image_data.name,
|
||||
]
|
||||
# ACT
|
||||
name_list = image_repo.list_names()
|
||||
# ASSERT
|
||||
assert len(name_list) == 2
|
||||
for name in name_list:
|
||||
assert name in expected_name_list
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
pytest.main(['-s', '-v', __file__])
|
||||
@@ -0,0 +1,133 @@
|
||||
"""Integration tests related to Minio implementation."""
|
||||
|
||||
import logging
|
||||
import os
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from shared.repositories.src.implementations import MinioImplementation
|
||||
|
||||
|
||||
def same_data(
|
||||
data_a: BytesIO,
|
||||
data_b: BytesIO,
|
||||
) -> bool:
|
||||
"""Check if two BytesIO-objects contain the same data."""
|
||||
assert isinstance(data_a, BytesIO)
|
||||
assert isinstance(data_b, BytesIO)
|
||||
# prepare for being read
|
||||
data_a.seek(0)
|
||||
data_b.seek(0)
|
||||
# convert to bytes
|
||||
data_a_bytes = data_a.read()
|
||||
data_b_bytes = data_b.read()
|
||||
# compare size
|
||||
if len(data_a_bytes) != len(data_b_bytes):
|
||||
logging.error(
|
||||
'data has different length: %s and %s',
|
||||
len(data_a_bytes),
|
||||
len(data_b_bytes),
|
||||
)
|
||||
return False
|
||||
# compare content
|
||||
if data_a_bytes != data_b_bytes:
|
||||
logging.error('data has different bytes')
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def test_should_connect_to_minio():
|
||||
"""Test connection to Minio."""
|
||||
# ARRANGE
|
||||
client = MinioImplementation()
|
||||
# ACT
|
||||
client.connect()
|
||||
# ASSERT
|
||||
assert client.connected()
|
||||
client.close()
|
||||
|
||||
|
||||
def test_should_have_context_handler():
|
||||
"""Test that class has a working context handler implemented."""
|
||||
# ACT
|
||||
with MinioImplementation() as client:
|
||||
# ASSERT
|
||||
assert client.connected()
|
||||
|
||||
|
||||
def test_should_get_data(
|
||||
minio_client: MinioImplementation,
|
||||
buffer_in_minio: tuple[Path, BytesIO],
|
||||
):
|
||||
"""Test getting data from Minio."""
|
||||
# ARRANGE
|
||||
path, buffer = buffer_in_minio
|
||||
# ACT
|
||||
received_buffer = minio_client._get(path)
|
||||
# ASSERT
|
||||
assert received_buffer is not None
|
||||
assert same_data(received_buffer, buffer)
|
||||
|
||||
|
||||
def test_should_get_none_when_no_data(
|
||||
minio_client: MinioImplementation,
|
||||
):
|
||||
"""Test getting None when no data is available in Minio."""
|
||||
# ARRANGE
|
||||
nonexistent_path = Path('nonexistent-object-name')
|
||||
# ACT
|
||||
received_buffer = minio_client._get(nonexistent_path)
|
||||
# ASSERT
|
||||
assert received_buffer is None
|
||||
|
||||
|
||||
def test_should_delete_data(
|
||||
minio_client: MinioImplementation,
|
||||
buffer_in_minio: tuple[Path, BytesIO],
|
||||
):
|
||||
"""Test deleting data from Minio."""
|
||||
# ARRANGE
|
||||
path, _ = buffer_in_minio
|
||||
# ACT
|
||||
minio_client._delete(path)
|
||||
# ASSERT
|
||||
received_buffer = minio_client._get(path)
|
||||
assert received_buffer is None
|
||||
|
||||
|
||||
def test_should_put_data(
|
||||
minio_client: MinioImplementation,
|
||||
buffer: BytesIO,
|
||||
):
|
||||
"""Test putting data in Minio."""
|
||||
# ARRANGE
|
||||
path = Path(os.getenv('MINIO_OBJECT_NAME', default=''))
|
||||
# ACT
|
||||
minio_client._put(path, buffer)
|
||||
received_buffer = minio_client._get(path)
|
||||
# ASSERT
|
||||
assert received_buffer is not None
|
||||
assert same_data(received_buffer, buffer)
|
||||
|
||||
|
||||
def test_should_update_data(
|
||||
minio_client: MinioImplementation,
|
||||
buffer_in_minio: tuple[Path, BytesIO],
|
||||
):
|
||||
"""Test updating data in Minio."""
|
||||
# ARRANGE
|
||||
path, buffer = buffer_in_minio
|
||||
updated_buffer = BytesIO(buffer.getvalue() + b'extra data')
|
||||
# ACT
|
||||
minio_client._put(path, updated_buffer)
|
||||
received_buffer = minio_client._get(path)
|
||||
# ASSERT
|
||||
assert not same_data(updated_buffer, buffer)
|
||||
assert received_buffer is not None
|
||||
assert same_data(received_buffer, updated_buffer)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
pytest.main(['-s', '-v', __file__])
|
||||
@@ -0,0 +1,136 @@
|
||||
"""Integration tests for ModelRepository class."""
|
||||
|
||||
from io import BytesIO
|
||||
|
||||
import pytest
|
||||
|
||||
from shared.repositories import ModelData, ModelRepository
|
||||
|
||||
|
||||
def same_buffer(
|
||||
buffer_a: BytesIO,
|
||||
buffer_b: BytesIO,
|
||||
) -> bool:
|
||||
"""Check if 2 buffers contain the same data."""
|
||||
assert isinstance(buffer_a, BytesIO)
|
||||
assert isinstance(buffer_b, BytesIO)
|
||||
# read buffers
|
||||
a_values = buffer_a.getvalue()
|
||||
b_values = buffer_b.getvalue()
|
||||
# compare length of buffers
|
||||
if len(a_values) != len(b_values):
|
||||
return False
|
||||
# compare content of buffers
|
||||
if a_values != b_values:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def same_model_data(
|
||||
data_a: ModelData,
|
||||
data_b: ModelData,
|
||||
) -> bool:
|
||||
"""Check if to ModelData-objects contain the same data."""
|
||||
assert isinstance(data_a, ModelData)
|
||||
assert isinstance(data_b, ModelData)
|
||||
# compare names
|
||||
if data_a.buffer_checksum != data_b.buffer_checksum:
|
||||
return False
|
||||
# compare buffer
|
||||
if not same_buffer(data_a.buffer, data_b.buffer):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def test_should_have_context_handler():
|
||||
"""Test that class has a working context handler implemented."""
|
||||
# ACT
|
||||
with ModelRepository() as repo:
|
||||
# ASSERT
|
||||
assert repo.connected()
|
||||
|
||||
|
||||
def test_should_get_model_data(
|
||||
model_repo: ModelRepository,
|
||||
model_data_in_minio: ModelData,
|
||||
):
|
||||
"""Test getting model data."""
|
||||
# ARRANGE
|
||||
object_name = ModelRepository._build_object_name(model_data_in_minio)
|
||||
# ACT
|
||||
received_model_data = model_repo.get_data(object_name)
|
||||
# ASSERT
|
||||
assert received_model_data is not None
|
||||
assert isinstance(received_model_data, ModelData)
|
||||
assert same_model_data(model_data_in_minio, received_model_data)
|
||||
|
||||
|
||||
def test_should_get_none_when_no_model_data(
|
||||
model_repo: ModelRepository,
|
||||
model_data: ModelData,
|
||||
):
|
||||
"""Test getting None whne no data is available."""
|
||||
# ARRANGE
|
||||
object_name = ModelRepository._build_object_name(model_data)
|
||||
# ACT
|
||||
received_model_data = model_repo.get_data(object_name)
|
||||
# ASSERT
|
||||
assert received_model_data is None
|
||||
|
||||
|
||||
def test_should_delete_model_data(
|
||||
model_repo: ModelRepository,
|
||||
model_data_in_minio: ModelData,
|
||||
):
|
||||
"""Test deleting model data."""
|
||||
# ARRANGE
|
||||
object_name = ModelRepository._build_object_name(model_data_in_minio)
|
||||
# ACT
|
||||
model_repo.remove_data(object_name)
|
||||
received_model_data = model_repo.get_data(object_name)
|
||||
# ASSERT
|
||||
assert received_model_data is None
|
||||
|
||||
|
||||
def test_should_put_model_data(
|
||||
model_repo: ModelRepository,
|
||||
model_data: ModelData,
|
||||
):
|
||||
"""Test putting model data."""
|
||||
# ARRANGE
|
||||
object_name = ModelRepository._build_object_name(model_data)
|
||||
# ACT
|
||||
model_repo.put_data(model_data)
|
||||
received_model_data = model_repo.get_data(object_name)
|
||||
# ASSERT
|
||||
assert received_model_data is not None
|
||||
assert same_model_data(model_data, received_model_data)
|
||||
|
||||
|
||||
def test_should_list_names(
|
||||
model_repo: ModelRepository,
|
||||
model_data_in_minio: ModelData,
|
||||
):
|
||||
"""Test get all model names."""
|
||||
# ARRANGE
|
||||
new_buffer_checksum = ModelData.calculate_checksum(model_data_in_minio.buffer)
|
||||
updated_model_data = model_data_in_minio.model_copy(
|
||||
update={
|
||||
'buffer_checksum': new_buffer_checksum,
|
||||
},
|
||||
)
|
||||
model_repo.put_data(updated_model_data)
|
||||
expected_name_list = [
|
||||
ModelRepository._build_object_name(model_data_in_minio),
|
||||
ModelRepository._build_object_name(updated_model_data),
|
||||
]
|
||||
# ACT
|
||||
name_list = model_repo.list_names()
|
||||
# ASSERT
|
||||
assert len(name_list) == 2
|
||||
for name in name_list:
|
||||
assert name in expected_name_list
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
pytest.main(['-s', '-v', __file__])
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user