Many of my duties and events are more or less sorted out, my concentration on drawing is finally back under this cooled down weather. I am planning to scan this image to colour with CLIP STUDIO PAINT later.
I have asked ChatGPT plus to generate this python programming code to transform an image file (PDF, JPG, PNG) scanned by a smart phone to a PNG file with transparency.
In this way, this scanned image can be scanned to a digital painting tool to use it as the base line.
# ============================================================
# Pencil Drawing to Transparent PNG Converter - Version 2
# Google Colab
#
# Supported input:
# - PDF
# - JPG / JPEG
# - PNG
#
# Output:
# - Transparent PNG
#
# Processing policy:
# 1. Remove white / near-white paper background.
# 2. Make the paper background truly transparent.
# 3. Preserve pencil-line intensity using the alpha channel.
# 4. Store all pencil lines as pure black RGB.
# 5. Remove photographed areas outside the paper.
# 6. Do not redraw, sharpen, denoise, or stylize the artwork.
# 7. Do not rotate or perspective-warp the original artwork.
# 8. Automatically limit PDF rendering size to avoid
# PyMuPDF "Overly large image" errors.
# ============================================================
# ------------------------------------------------------------
# Install required packages
# ------------------------------------------------------------
import sys
import subprocess
import importlib.util
required_packages = {
"fitz": "PyMuPDF",
"cv2": "opencv-python-headless",
"PIL": "Pillow",
}
for module_name, package_name in required_packages.items():
if importlib.util.find_spec(module_name) is None:
subprocess.check_call(
[
sys.executable,
"-m",
"pip",
"install",
"-q",
package_name,
]
)
# ------------------------------------------------------------
# Imports
# ------------------------------------------------------------
import os
import io
import math
import zipfile
import numpy as np
import cv2
import fitz
from PIL import Image, ImageOps, ImageDraw
from google.colab import files
from IPython.display import display
# ============================================================
# USER SETTINGS
# ============================================================
# Preferred PDF rendering resolution.
#
# 300 DPI is recommended for normal pencil drawings.
# The program automatically reduces the effective DPI only
# when the rendered page would otherwise become too large.
PDF_DPI = 300
# Maximum number of pixels allowed for one rendered PDF page.
#
# This prevents:
# FzErrorLimit: Overly large image
#
# 25 million pixels is a good balance for Google Colab.
MAX_PDF_PIXELS = 25_000_000
# Automatically crop obvious photographed borders
# outside the white sheet of paper.
AUTO_CROP_PAPER = True
# Make areas outside a slightly tilted sheet transparent.
#
# This does NOT rotate or warp the drawing.
REMOVE_OUTSIDE_PAPER = True
# Small paper-noise suppression value.
#
# 0 = preserve absolutely everything, including paper noise.
# 1-3 = recommended for pencil drawings.
# Larger values remove more paper texture but can also remove
# extremely faint pencil construction lines.
BACKGROUND_NOISE_FLOOR = 2
# White-reference percentile.
#
# High percentile values estimate the true paper white.
WHITE_PERCENTILE = 99.0
# Pencil strength curve.
#
# 1.00 = preserve the original tonal relationship.
# <1.00 = make faint pencil lines slightly more visible.
# >1.00 = make faint lines weaker.
#
# Recommended starting value:
LINE_GAMMA = 1.00
# Minimum area ratio required for automatic paper detection.
MIN_PAPER_AREA_RATIO = 0.30
# Maximum image dimension used internally for paper detection.
# This does not change the final image resolution.
DETECTION_MAX_SIDE = 1600
# Show a checkerboard preview in Colab.
SHOW_PREVIEW = True
# Automatically download the finished output.
AUTO_DOWNLOAD = True
# ============================================================
# HELPER: LONGEST CONTINUOUS REGION
# ============================================================
def longest_true_run(mask):
"""
Return the start and end indices of the longest
continuous True region.
"""
best_start = 0
best_end = len(mask) - 1
best_length = 0
current_start = None
for i, value in enumerate(mask):
if value and current_start is None:
current_start = i
if current_start is not None:
is_last = (i == len(mask) - 1)
if (not value) or is_last:
if value and is_last:
end = i
else:
end = i - 1
length = end - current_start + 1
if length > best_length:
best_start = current_start
best_end = end
best_length = length
current_start = None
return best_start, best_end
# ============================================================
# PAPER BORDER DETECTION
# ============================================================
def detect_paper_bbox(rgb):
"""
Detect obvious dark photographed borders around the paper.
Only an axis-aligned crop is performed.
No rotation, perspective correction, or geometric
transformation is applied.
"""
gray = cv2.cvtColor(
rgb,
cv2.COLOR_RGB2GRAY
)
white_reference = np.percentile(
gray,
95
)
edge_threshold = max(
160,
int(white_reference - 35)
)
row_median = np.median(
gray,
axis=1
)
col_median = np.median(
gray,
axis=0
)
valid_rows = (
row_median >= edge_threshold
)
valid_cols = (
col_median >= edge_threshold
)
y0, y1 = longest_true_run(
valid_rows
)
x0, x1 = longest_true_run(
valid_cols
)
height, width = gray.shape
detected_area = (
(x1 - x0 + 1)
*
(y1 - y0 + 1)
)
full_area = width * height
if (
detected_area
<
full_area * MIN_PAPER_AREA_RATIO
):
return 0, 0, width, height
x0 = max(0, x0)
y0 = max(0, y0)
x1 = min(
width,
x1 + 1
)
y1 = min(
height,
y1 + 1
)
return x0, y0, x1, y1
# ============================================================
# PAPER-SHAPE MASK
# ============================================================
def detect_paper_mask(rgb):
"""
Detect the approximate shape of the sheet of paper.
The detected shape is used only for transparency.
The artwork itself is never geometrically transformed.
"""
height, width = rgb.shape[:2]
scale = min(
1.0,
DETECTION_MAX_SIDE
/
float(max(height, width))
)
if scale < 1.0:
small_width = max(
1,
int(round(width * scale))
)
small_height = max(
1,
int(round(height * scale))
)
small = cv2.resize(
rgb,
(small_width, small_height),
interpolation=cv2.INTER_AREA
)
else:
small = rgb.copy()
gray = cv2.cvtColor(
small,
cv2.COLOR_RGB2GRAY
)
gray = cv2.GaussianBlur(
gray,
(5, 5),
0
)
white_reference = np.percentile(
gray,
90
)
paper_threshold = max(
140,
min(
245,
int(white_reference - 45)
)
)
bright_mask = np.where(
gray >= paper_threshold,
255,
0
).astype(np.uint8)
kernel_size = max(
5,
int(
min(
bright_mask.shape
)
* 0.02
)
)
if kernel_size % 2 == 0:
kernel_size += 1
kernel = cv2.getStructuringElement(
cv2.MORPH_RECT,
(kernel_size, kernel_size)
)
bright_mask = cv2.morphologyEx(
bright_mask,
cv2.MORPH_CLOSE,
kernel,
iterations=2
)
contours, _ = cv2.findContours(
bright_mask,
cv2.RETR_EXTERNAL,
cv2.CHAIN_APPROX_SIMPLE
)
if not contours:
return np.full(
(height, width),
255,
dtype=np.uint8
)
largest_contour = max(
contours,
key=cv2.contourArea
)
contour_area = cv2.contourArea(
largest_contour
)
image_area = (
bright_mask.shape[0]
*
bright_mask.shape[1]
)
if (
contour_area
<
image_area * MIN_PAPER_AREA_RATIO
):
return np.full(
(height, width),
255,
dtype=np.uint8
)
hull = cv2.convexHull(
largest_contour
)
hull = hull[:, 0, :].astype(
np.float32
)
hull /= scale
hull = np.rint(
hull
).astype(np.int32)
paper_mask = np.zeros(
(height, width),
dtype=np.uint8
)
cv2.fillConvexPoly(
paper_mask,
hull,
255
)
return paper_mask
# ============================================================
# BLACK PENCIL + TRANSPARENT PAPER
# ============================================================
def create_black_pencil_alpha(
rgb,
original_alpha,
paper_mask
):
"""
Convert the scanned pencil drawing into:
RGB = pure black
Alpha = original pencil darkness
White paper becomes Alpha = 0.
This is preferable to preserving light-gray RGB values
because the final pencil color remains black while
original pencil intensity is represented by opacity.
When placed over a white background, the visual tonal
relationship remains close to the original scan.
"""
gray = cv2.cvtColor(
rgb,
cv2.COLOR_RGB2GRAY
).astype(np.float32)
paper_pixels = gray[
paper_mask > 0
]
if paper_pixels.size == 0:
paper_pixels = gray.reshape(
-1
)
white_reference = float(
np.percentile(
paper_pixels,
WHITE_PERCENTILE
)
)
white_reference = float(
np.clip(
white_reference,
180.0,
255.0
)
)
# Calculate darkness relative to the detected paper white.
darkness = (
white_reference
-
gray
-
BACKGROUND_NOISE_FLOOR
)
darkness = np.clip(
darkness,
0.0,
white_reference
)
darkness /= white_reference
# Optional tonal adjustment.
if LINE_GAMMA != 1.0:
darkness = np.power(
darkness,
LINE_GAMMA
)
alpha = np.rint(
darkness * 255.0
).astype(np.uint8)
# Preserve transparency from an already-transparent PNG.
alpha = np.minimum(
alpha,
original_alpha
)
# Remove anything detected outside the paper.
if REMOVE_OUTSIDE_PAPER:
alpha[
paper_mask == 0
] = 0
print(
f"Detected paper-white level: "
f"{white_reference:.1f}"
)
return alpha
# ============================================================
# PROCESS ONE IMAGE
# ============================================================
def process_pil_image(pil_image):
"""
Process one input image and return a transparent RGBA PNG.
"""
pil_image = ImageOps.exif_transpose(
pil_image
)
rgba = np.array(
pil_image.convert("RGBA")
)
rgb = rgba[:, :, :3].copy()
original_alpha = rgba[
:,
:,
3
].copy()
# --------------------------------------------------------
# Crop photographed borders outside the paper
# --------------------------------------------------------
if AUTO_CROP_PAPER:
x0, y0, x1, y1 = detect_paper_bbox(
rgb
)
rgb = rgb[
y0:y1,
x0:x1
]
original_alpha = original_alpha[
y0:y1,
x0:x1
]
print(
"Paper crop:"
f" x={x0}:{x1},"
f" y={y0}:{y1}"
)
# --------------------------------------------------------
# Detect the paper shape
# --------------------------------------------------------
if REMOVE_OUTSIDE_PAPER:
paper_mask = detect_paper_mask(
rgb
)
else:
paper_mask = np.full(
rgb.shape[:2],
255,
dtype=np.uint8
)
# --------------------------------------------------------
# Create alpha from pencil darkness
# --------------------------------------------------------
alpha = create_black_pencil_alpha(
rgb,
original_alpha,
paper_mask
)
height, width = alpha.shape
# --------------------------------------------------------
# Output RGB is pure black.
#
# Pencil darkness is preserved using Alpha.
# Background remains transparent.
# --------------------------------------------------------
output_rgba = np.zeros(
(height, width, 4),
dtype=np.uint8
)
output_rgba[:, :, 0] = 0
output_rgba[:, :, 1] = 0
output_rgba[:, :, 2] = 0
output_rgba[:, :, 3] = alpha
transparent_ratio = (
np.mean(alpha == 0)
*
100.0
)
print(
f"Fully transparent pixels: "
f"{transparent_ratio:.1f}%"
)
return Image.fromarray(
output_rgba,
mode="RGBA"
)
# ============================================================
# SAFE PDF RENDERING
# ============================================================
def render_pdf_page_safely(
page,
page_number
):
"""
Render one PDF page while automatically limiting
the pixel count.
This prevents:
FzErrorLimit: Overly large image
"""
width_points = float(
page.rect.width
)
height_points = float(
page.rect.height
)
desired_scale = (
PDF_DPI / 72.0
)
estimated_width = (
width_points
*
desired_scale
)
estimated_height = (
height_points
*
desired_scale
)
estimated_pixels = (
estimated_width
*
estimated_height
)
scale = desired_scale
if estimated_pixels > MAX_PDF_PIXELS:
scale *= math.sqrt(
MAX_PDF_PIXELS
/
estimated_pixels
)
# Retry with progressively smaller dimensions if
# PyMuPDF still reports an oversized pixmap.
last_error = None
for attempt in range(6):
effective_dpi = (
scale * 72.0
)
print(
f"Rendering PDF page "
f"{page_number + 1}: "
f"{effective_dpi:.1f} DPI"
)
try:
matrix = fitz.Matrix(
scale,
scale
)
pixmap = page.get_pixmap(
matrix=matrix,
colorspace=fitz.csRGB,
alpha=False
)
image = Image.frombytes(
"RGB",
(
pixmap.width,
pixmap.height
),
pixmap.samples
)
print(
f"Rendered size: "
f"{pixmap.width} x "
f"{pixmap.height}"
)
return image
except Exception as error:
last_error = error
print(
"PDF render was too large. "
"Retrying at a smaller size..."
)
scale *= 0.75
raise RuntimeError(
"Unable to render PDF page safely. "
f"Last error: {last_error}"
)
def render_pdf(pdf_data):
"""
Render every PDF page safely.
"""
document = fitz.open(
stream=pdf_data,
filetype="pdf"
)
images = []
try:
for page_number in range(
len(document)
):
page = document.load_page(
page_number
)
image = render_pdf_page_safely(
page,
page_number
)
images.append(
image
)
finally:
document.close()
return images
# ============================================================
# CHECKERBOARD TRANSPARENCY PREVIEW
# ============================================================
def make_checkerboard_preview(
rgba_image,
checker_size=24
):
"""
Create a checkerboard preview so transparent pixels
are visibly distinguishable from black pixels.
This preview is NOT saved to the output file.
"""
rgba_image = rgba_image.convert(
"RGBA"
)
width, height = rgba_image.size
background = Image.new(
"RGB",
(width, height),
(230, 230, 230)
)
draw = ImageDraw.Draw(
background
)
second_color = (
190,
190,
190
)
for y in range(
0,
height,
checker_size
):
for x in range(
0,
width,
checker_size
):
if (
(
x // checker_size
+
y // checker_size
)
%
2
==
1
):
draw.rectangle(
[
x,
y,
min(
x + checker_size - 1,
width - 1
),
min(
y + checker_size - 1,
height - 1
),
],
fill=second_color
)
background = background.convert(
"RGBA"
)
preview = Image.alpha_composite(
background,
rgba_image
)
return preview
# ============================================================
# FILE UPLOAD
# ============================================================
print(
"Upload PDF, JPG, JPEG, or PNG files."
)
uploaded = files.upload()
# ============================================================
# PROCESS FILES
# ============================================================
output_files = []
for filename, file_data in uploaded.items():
extension = os.path.splitext(
filename
)[1].lower()
base_name = os.path.splitext(
os.path.basename(filename)
)[0]
print()
print("=" * 70)
print(
f"Processing: {filename}"
)
print("=" * 70)
images = []
# --------------------------------------------------------
# PDF
# --------------------------------------------------------
if extension == ".pdf":
images = render_pdf(
file_data
)
# --------------------------------------------------------
# JPG / JPEG / PNG
# --------------------------------------------------------
elif extension in [
".jpg",
".jpeg",
".png"
]:
image = Image.open(
io.BytesIO(
file_data
)
)
images = [
image
]
else:
print(
f"Unsupported file type: "
f"{filename}"
)
continue
# --------------------------------------------------------
# Process each image / PDF page
# --------------------------------------------------------
for index, image in enumerate(
images
):
print()
print(
f"Processing page/image "
f"{index + 1}..."
)
result = process_pil_image(
image
)
if extension == ".pdf":
output_name = (
f"{base_name}"
f"_page_"
f"{index + 1:03d}"
f"_transparent.png"
)
else:
output_name = (
f"{base_name}"
f"_transparent.png"
)
result.save(
output_name,
format="PNG"
)
output_files.append(
output_name
)
print(
f"Saved: {output_name}"
)
# ----------------------------------------------------
# Show transparency using checkerboard preview
# ----------------------------------------------------
if SHOW_PREVIEW:
print(
"Preview:"
" checkerboard = transparent area"
)
preview = make_checkerboard_preview(
result
)
display(
preview
)
# ============================================================
# DOWNLOAD RESULTS
# ============================================================
if len(output_files) == 1:
print()
print(
"Finished successfully."
)
print(
f"Output: {output_files[0]}"
)
if AUTO_DOWNLOAD:
files.download(
output_files[0]
)
elif len(output_files) > 1:
zip_name = (
"transparent_png_results.zip"
)
with zipfile.ZipFile(
zip_name,
"w",
compression=zipfile.ZIP_DEFLATED
) as zip_file:
for output_file in output_files:
zip_file.write(
output_file,
arcname=os.path.basename(
output_file
)
)
print()
print(
f"Finished successfully."
)
print(
f"{len(output_files)} files "
f"were saved to {zip_name}."
)
if AUTO_DOWNLOAD:
files.download(
zip_name
)
else:
print(
"No output files were created."
)
_page_001_transparent.png)
