-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathDockerfile
More file actions
156 lines (125 loc) · 6.26 KB
/
Copy pathDockerfile
File metadata and controls
156 lines (125 loc) · 6.26 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
# ============================================================================
# Meter OCR pipeline - one Dockerfile, four targets.
#
# docker build -t meter-ocr . # default: both engines, CPU
# docker build --target slim -t meter-ocr:slim . # easyocr only, smaller
# docker build --target gpu -t meter-ocr:gpu . # easyocr on CUDA 12.4
#
# Run it 24/7, reading every image you drop into .\images :
#
# docker run -d --name meter-ocr --restart unless-stopped ^
# -v "%cd%\images:/app/images:ro" ^
# -v "%cd%\data:/app/data" ^
# -v "%cd%\logs:/app/logs" ^
# -v "%cd%\config.yaml:/app/config.yaml:ro" ^
# meter-ocr
#
# Everything else is a subcommand on the same image - see README.md.
# ============================================================================
# ---------------------------------------------------------------------------
# base - system packages and the dependencies every target shares
# ---------------------------------------------------------------------------
FROM python:3.10-slim AS base
# ffmpeg - RTSP capture goes through cv2.CAP_FFMPEG
# libglib2.0-0 - still required by opencv-python-headless
# ca-certificates - TLS for the one-time model downloads below
RUN apt-get update && apt-get install -y --no-install-recommends \
ffmpeg \
libglib2.0-0 \
ca-certificates \
&& rm -rf /var/lib/apt/lists/*
ENV PYTHONUNBUFFERED=1 \
PYTHONDONTWRITEBYTECODE=1 \
PIP_NO_CACHE_DIR=1 \
TZ=Asia/Kolkata
WORKDIR /app
# ---------------------------------------------------------------------------
# easyocr-core - torch + easyocr with the weights baked in.
# Shared by the `cpu` and `both` targets so neither pays for it twice.
# ---------------------------------------------------------------------------
FROM base AS easyocr-core
# torch from the CPU index. The default PyPI wheel drags in the entire CUDA
# runtime (~2.5GB) even when it will never touch a GPU.
RUN pip install --no-cache-dir \
torch==2.6.0 torchvision==0.21.0 \
--index-url https://download.pytorch.org/whl/cpu
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
# Bake the EasyOCR weights (~100MB) in. Without this, every fresh container
# downloads them on first read - which breaks air-gapped deployment and makes
# the first reading silently slow.
RUN python -c "import easyocr; easyocr.Reader(['en'], gpu=False)"
# ---------------------------------------------------------------------------
# gpu - EasyOCR on CUDA 12.4. Needs `--gpus all` at run time AND
# `gpu: true` in config.yaml; miss either and torch quietly uses the CPU.
#
# NOTE: this target ships EasyOCR only, so it also needs `engine: easyocr`
# in config.yaml - which contradicts the measured default. GPU acceleration
# for paddle needs the paddlepaddle-gpu wheel, which is a different package
# built against its own CUDA. Add it here if you need paddle on the GPU.
# ---------------------------------------------------------------------------
FROM nvidia/cuda:12.4.1-cudnn-runtime-ubuntu22.04 AS gpu
RUN apt-get update && apt-get install -y --no-install-recommends \
python3.10 python3-pip ffmpeg libglib2.0-0 ca-certificates \
&& rm -rf /var/lib/apt/lists/* \
&& ln -sf /usr/bin/python3.10 /usr/local/bin/python
ENV PYTHONUNBUFFERED=1 \
PYTHONDONTWRITEBYTECODE=1 \
PIP_NO_CACHE_DIR=1 \
TZ=Asia/Kolkata
WORKDIR /app
RUN python -m pip install --no-cache-dir --upgrade pip \
&& python -m pip install --no-cache-dir \
torch==2.6.0 torchvision==0.21.0 \
--index-url https://download.pytorch.org/whl/cu124
COPY requirements.txt .
RUN python -m pip install --no-cache-dir -r requirements.txt
# gpu=False here only chooses where the weights load for this throwaway call;
# the download path and the files on disk are identical.
RUN python -c "import easyocr; easyocr.Reader(['en'], gpu=False)"
COPY meter_ocr/ ./meter_ocr/
COPY config.yaml ./
RUN mkdir -p images data/snapshots data/annotated logs
ENTRYPOINT ["python", "-m", "meter_ocr"]
CMD ["watch"]
# ---------------------------------------------------------------------------
# slim - EasyOCR only, ~1.5GB smaller. Build this if you have measured
# easyocr to be good enough for your meter and want the smaller image.
# Requires `engine: easyocr` in config.yaml.
# ---------------------------------------------------------------------------
FROM easyocr-core AS slim
COPY meter_ocr/ ./meter_ocr/
COPY config.yaml ./
RUN mkdir -p images data/snapshots data/annotated logs
ENTRYPOINT ["python", "-m", "meter_ocr"]
CMD ["watch"]
# ---------------------------------------------------------------------------
# DEFAULT TARGET - both engines.
#
# Last stage in the file, so a bare `docker build .` produces this one. It
# carries easyocr AND paddleocr because on the meter photos in this project
# paddleocr is the better of the two (it localises the whole register and
# reads the "MD" page label at 0.98; easyocr finds only 2-3 characters), and
# because `compare` needs both present to be able to re-check that on your
# images. See README.md for the measured numbers.
# ---------------------------------------------------------------------------
FROM easyocr-core AS full
# Pinned as a pair: paddleocr 2.7.x is the last line whose .ocr(img, cls=True)
# call signature matches engines.PaddleOCREngine. 3.x reorganised the API.
RUN pip install --no-cache-dir paddlepaddle==2.6.2 paddleocr==2.7.3
# Bake the paddle weights in too, for the same reason as easyocr's: no
# first-run download, and the image works with no internet.
RUN python -c "from paddleocr import PaddleOCR; PaddleOCR(use_angle_cls=True, lang='en', show_log=False)"
# Code last: editing a .py rebuilds only this layer, leaving the torch install
# and both model downloads above cached.
COPY meter_ocr/ ./meter_ocr/
COPY config.yaml ./
# images/ is the drop folder; the rest is runtime output. All are normally
# bind-mounted over, but they must exist for the container to run with no
# mounts at all.
RUN mkdir -p images data/snapshots data/annotated logs
# The entrypoint is the CLI, so the subcommand is all you pass:
# docker run --rm meter-ocr selftest
# docker run --rm meter-ocr image images/meter_kwh.jpg
ENTRYPOINT ["python", "-m", "meter_ocr"]
CMD ["watch"]