Monday, October 05, 2026

Python: Scanned Pencil Drawing Background Removal and Transparency Converter

 Many of my duties and events are more or less sorted out, my concentration on drawing is finally back under this cooled down weather. I am planning to scan this image to colour with CLIP STUDIO PAINT later.

  I have asked ChatGPT plus to generate this python programming code to transform an image file (PDF, JPG, PNG) scanned by a smart phone to a PNG file with transparency. 

In this way, this scanned image can be scanned to a digital painting tool to use it as the base line. 

 

# ============================================================
# Pencil Drawing to Transparent PNG Converter - Version 2
# Google Colab
#
# Supported input:
#   - PDF
#   - JPG / JPEG
#   - PNG
#
# Output:
#   - Transparent PNG
#
# Processing policy:
#   1. Remove white / near-white paper background.
#   2. Make the paper background truly transparent.
#   3. Preserve pencil-line intensity using the alpha channel.
#   4. Store all pencil lines as pure black RGB.
#   5. Remove photographed areas outside the paper.
#   6. Do not redraw, sharpen, denoise, or stylize the artwork.
#   7. Do not rotate or perspective-warp the original artwork.
#   8. Automatically limit PDF rendering size to avoid
#      PyMuPDF "Overly large image" errors.
# ============================================================


# ------------------------------------------------------------
# Install required packages
# ------------------------------------------------------------

import sys
import subprocess
import importlib.util

required_packages = {
    "fitz": "PyMuPDF",
    "cv2": "opencv-python-headless",
    "PIL": "Pillow",
}

for module_name, package_name in required_packages.items():
    if importlib.util.find_spec(module_name) is None:
        subprocess.check_call(
            [
                sys.executable,
                "-m",
                "pip",
                "install",
                "-q",
                package_name,
            ]
        )


# ------------------------------------------------------------
# Imports
# ------------------------------------------------------------

import os
import io
import math
import zipfile

import numpy as np
import cv2
import fitz

from PIL import Image, ImageOps, ImageDraw
from google.colab import files
from IPython.display import display


# ============================================================
# USER SETTINGS
# ============================================================

# Preferred PDF rendering resolution.
#
# 300 DPI is recommended for normal pencil drawings.
# The program automatically reduces the effective DPI only
# when the rendered page would otherwise become too large.
PDF_DPI = 300


# Maximum number of pixels allowed for one rendered PDF page.
#
# This prevents:
#   FzErrorLimit: Overly large image
#
# 25 million pixels is a good balance for Google Colab.
MAX_PDF_PIXELS = 25_000_000


# Automatically crop obvious photographed borders
# outside the white sheet of paper.
AUTO_CROP_PAPER = True


# Make areas outside a slightly tilted sheet transparent.
#
# This does NOT rotate or warp the drawing.
REMOVE_OUTSIDE_PAPER = True


# Small paper-noise suppression value.
#
# 0 = preserve absolutely everything, including paper noise.
# 1-3 = recommended for pencil drawings.
# Larger values remove more paper texture but can also remove
# extremely faint pencil construction lines.
BACKGROUND_NOISE_FLOOR = 2


# White-reference percentile.
#
# High percentile values estimate the true paper white.
WHITE_PERCENTILE = 99.0


# Pencil strength curve.
#
# 1.00 = preserve the original tonal relationship.
# <1.00 = make faint pencil lines slightly more visible.
# >1.00 = make faint lines weaker.
#
# Recommended starting value:
LINE_GAMMA = 1.00


# Minimum area ratio required for automatic paper detection.
MIN_PAPER_AREA_RATIO = 0.30


# Maximum image dimension used internally for paper detection.
# This does not change the final image resolution.
DETECTION_MAX_SIDE = 1600


# Show a checkerboard preview in Colab.
SHOW_PREVIEW = True


# Automatically download the finished output.
AUTO_DOWNLOAD = True


# ============================================================
# HELPER: LONGEST CONTINUOUS REGION
# ============================================================

def longest_true_run(mask):
    """
    Return the start and end indices of the longest
    continuous True region.
    """

    best_start = 0
    best_end = len(mask) - 1
    best_length = 0

    current_start = None

    for i, value in enumerate(mask):

        if value and current_start is None:
            current_start = i

        if current_start is not None:

            is_last = (i == len(mask) - 1)

            if (not value) or is_last:

                if value and is_last:
                    end = i
                else:
                    end = i - 1

                length = end - current_start + 1

                if length > best_length:
                    best_start = current_start
                    best_end = end
                    best_length = length

                current_start = None

    return best_start, best_end


# ============================================================
# PAPER BORDER DETECTION
# ============================================================

def detect_paper_bbox(rgb):
    """
    Detect obvious dark photographed borders around the paper.

    Only an axis-aligned crop is performed.
    No rotation, perspective correction, or geometric
    transformation is applied.
    """

    gray = cv2.cvtColor(
        rgb,
        cv2.COLOR_RGB2GRAY
    )

    white_reference = np.percentile(
        gray,
        95
    )

    edge_threshold = max(
        160,
        int(white_reference - 35)
    )

    row_median = np.median(
        gray,
        axis=1
    )

    col_median = np.median(
        gray,
        axis=0
    )

    valid_rows = (
        row_median >= edge_threshold
    )

    valid_cols = (
        col_median >= edge_threshold
    )

    y0, y1 = longest_true_run(
        valid_rows
    )

    x0, x1 = longest_true_run(
        valid_cols
    )

    height, width = gray.shape

    detected_area = (
        (x1 - x0 + 1)
        *
        (y1 - y0 + 1)
    )

    full_area = width * height

    if (
        detected_area
        <
        full_area * MIN_PAPER_AREA_RATIO
    ):
        return 0, 0, width, height

    x0 = max(0, x0)
    y0 = max(0, y0)

    x1 = min(
        width,
        x1 + 1
    )

    y1 = min(
        height,
        y1 + 1
    )

    return x0, y0, x1, y1


# ============================================================
# PAPER-SHAPE MASK
# ============================================================

def detect_paper_mask(rgb):
    """
    Detect the approximate shape of the sheet of paper.

    The detected shape is used only for transparency.
    The artwork itself is never geometrically transformed.
    """

    height, width = rgb.shape[:2]

    scale = min(
        1.0,
        DETECTION_MAX_SIDE
        /
        float(max(height, width))
    )

    if scale < 1.0:

        small_width = max(
            1,
            int(round(width * scale))
        )

        small_height = max(
            1,
            int(round(height * scale))
        )

        small = cv2.resize(
            rgb,
            (small_width, small_height),
            interpolation=cv2.INTER_AREA
        )

    else:

        small = rgb.copy()

    gray = cv2.cvtColor(
        small,
        cv2.COLOR_RGB2GRAY
    )

    gray = cv2.GaussianBlur(
        gray,
        (5, 5),
        0
    )

    white_reference = np.percentile(
        gray,
        90
    )

    paper_threshold = max(
        140,
        min(
            245,
            int(white_reference - 45)
        )
    )

    bright_mask = np.where(
        gray >= paper_threshold,
        255,
        0
    ).astype(np.uint8)

    kernel_size = max(
        5,
        int(
            min(
                bright_mask.shape
            )
            * 0.02
        )
    )

    if kernel_size % 2 == 0:
        kernel_size += 1

    kernel = cv2.getStructuringElement(
        cv2.MORPH_RECT,
        (kernel_size, kernel_size)
    )

    bright_mask = cv2.morphologyEx(
        bright_mask,
        cv2.MORPH_CLOSE,
        kernel,
        iterations=2
    )

    contours, _ = cv2.findContours(
        bright_mask,
        cv2.RETR_EXTERNAL,
        cv2.CHAIN_APPROX_SIMPLE
    )

    if not contours:

        return np.full(
            (height, width),
            255,
            dtype=np.uint8
        )

    largest_contour = max(
        contours,
        key=cv2.contourArea
    )

    contour_area = cv2.contourArea(
        largest_contour
    )

    image_area = (
        bright_mask.shape[0]
        *
        bright_mask.shape[1]
    )

    if (
        contour_area
        <
        image_area * MIN_PAPER_AREA_RATIO
    ):

        return np.full(
            (height, width),
            255,
            dtype=np.uint8
        )

    hull = cv2.convexHull(
        largest_contour
    )

    hull = hull[:, 0, :].astype(
        np.float32
    )

    hull /= scale

    hull = np.rint(
        hull
    ).astype(np.int32)

    paper_mask = np.zeros(
        (height, width),
        dtype=np.uint8
    )

    cv2.fillConvexPoly(
        paper_mask,
        hull,
        255
    )

    return paper_mask


# ============================================================
# BLACK PENCIL + TRANSPARENT PAPER
# ============================================================

def create_black_pencil_alpha(
    rgb,
    original_alpha,
    paper_mask
):
    """
    Convert the scanned pencil drawing into:

        RGB   = pure black
        Alpha = original pencil darkness

    White paper becomes Alpha = 0.

    This is preferable to preserving light-gray RGB values
    because the final pencil color remains black while
    original pencil intensity is represented by opacity.

    When placed over a white background, the visual tonal
    relationship remains close to the original scan.
    """

    gray = cv2.cvtColor(
        rgb,
        cv2.COLOR_RGB2GRAY
    ).astype(np.float32)

    paper_pixels = gray[
        paper_mask > 0
    ]

    if paper_pixels.size == 0:

        paper_pixels = gray.reshape(
            -1
        )

    white_reference = float(
        np.percentile(
            paper_pixels,
            WHITE_PERCENTILE
        )
    )

    white_reference = float(
        np.clip(
            white_reference,
            180.0,
            255.0
        )
    )

    # Calculate darkness relative to the detected paper white.
    darkness = (
        white_reference
        -
        gray
        -
        BACKGROUND_NOISE_FLOOR
    )

    darkness = np.clip(
        darkness,
        0.0,
        white_reference
    )

    darkness /= white_reference

    # Optional tonal adjustment.
    if LINE_GAMMA != 1.0:

        darkness = np.power(
            darkness,
            LINE_GAMMA
        )

    alpha = np.rint(
        darkness * 255.0
    ).astype(np.uint8)

    # Preserve transparency from an already-transparent PNG.
    alpha = np.minimum(
        alpha,
        original_alpha
    )

    # Remove anything detected outside the paper.
    if REMOVE_OUTSIDE_PAPER:

        alpha[
            paper_mask == 0
        ] = 0

    print(
        f"Detected paper-white level: "
        f"{white_reference:.1f}"
    )

    return alpha


# ============================================================
# PROCESS ONE IMAGE
# ============================================================

def process_pil_image(pil_image):
    """
    Process one input image and return a transparent RGBA PNG.
    """

    pil_image = ImageOps.exif_transpose(
        pil_image
    )

    rgba = np.array(
        pil_image.convert("RGBA")
    )

    rgb = rgba[:, :, :3].copy()

    original_alpha = rgba[
        :,
        :,
        3
    ].copy()

    # --------------------------------------------------------
    # Crop photographed borders outside the paper
    # --------------------------------------------------------

    if AUTO_CROP_PAPER:

        x0, y0, x1, y1 = detect_paper_bbox(
            rgb
        )

        rgb = rgb[
            y0:y1,
            x0:x1
        ]

        original_alpha = original_alpha[
            y0:y1,
            x0:x1
        ]

        print(
            "Paper crop:"
            f" x={x0}:{x1},"
            f" y={y0}:{y1}"
        )

    # --------------------------------------------------------
    # Detect the paper shape
    # --------------------------------------------------------

    if REMOVE_OUTSIDE_PAPER:

        paper_mask = detect_paper_mask(
            rgb
        )

    else:

        paper_mask = np.full(
            rgb.shape[:2],
            255,
            dtype=np.uint8
        )

    # --------------------------------------------------------
    # Create alpha from pencil darkness
    # --------------------------------------------------------

    alpha = create_black_pencil_alpha(
        rgb,
        original_alpha,
        paper_mask
    )

    height, width = alpha.shape

    # --------------------------------------------------------
    # Output RGB is pure black.
    #
    # Pencil darkness is preserved using Alpha.
    # Background remains transparent.
    # --------------------------------------------------------

    output_rgba = np.zeros(
        (height, width, 4),
        dtype=np.uint8
    )

    output_rgba[:, :, 0] = 0
    output_rgba[:, :, 1] = 0
    output_rgba[:, :, 2] = 0

    output_rgba[:, :, 3] = alpha

    transparent_ratio = (
        np.mean(alpha == 0)
        *
        100.0
    )

    print(
        f"Fully transparent pixels: "
        f"{transparent_ratio:.1f}%"
    )

    return Image.fromarray(
        output_rgba,
        mode="RGBA"
    )


# ============================================================
# SAFE PDF RENDERING
# ============================================================

def render_pdf_page_safely(
    page,
    page_number
):
    """
    Render one PDF page while automatically limiting
    the pixel count.

    This prevents:
        FzErrorLimit: Overly large image
    """

    width_points = float(
        page.rect.width
    )

    height_points = float(
        page.rect.height
    )

    desired_scale = (
        PDF_DPI / 72.0
    )

    estimated_width = (
        width_points
        *
        desired_scale
    )

    estimated_height = (
        height_points
        *
        desired_scale
    )

    estimated_pixels = (
        estimated_width
        *
        estimated_height
    )

    scale = desired_scale

    if estimated_pixels > MAX_PDF_PIXELS:

        scale *= math.sqrt(
            MAX_PDF_PIXELS
            /
            estimated_pixels
        )

    # Retry with progressively smaller dimensions if
    # PyMuPDF still reports an oversized pixmap.
    last_error = None

    for attempt in range(6):

        effective_dpi = (
            scale * 72.0
        )

        print(
            f"Rendering PDF page "
            f"{page_number + 1}: "
            f"{effective_dpi:.1f} DPI"
        )

        try:

            matrix = fitz.Matrix(
                scale,
                scale
            )

            pixmap = page.get_pixmap(
                matrix=matrix,
                colorspace=fitz.csRGB,
                alpha=False
            )

            image = Image.frombytes(
                "RGB",
                (
                    pixmap.width,
                    pixmap.height
                ),
                pixmap.samples
            )

            print(
                f"Rendered size: "
                f"{pixmap.width} x "
                f"{pixmap.height}"
            )

            return image

        except Exception as error:

            last_error = error

            print(
                "PDF render was too large. "
                "Retrying at a smaller size..."
            )

            scale *= 0.75

    raise RuntimeError(
        "Unable to render PDF page safely. "
        f"Last error: {last_error}"
    )


def render_pdf(pdf_data):
    """
    Render every PDF page safely.
    """

    document = fitz.open(
        stream=pdf_data,
        filetype="pdf"
    )

    images = []

    try:

        for page_number in range(
            len(document)
        ):

            page = document.load_page(
                page_number
            )

            image = render_pdf_page_safely(
                page,
                page_number
            )

            images.append(
                image
            )

    finally:

        document.close()

    return images


# ============================================================
# CHECKERBOARD TRANSPARENCY PREVIEW
# ============================================================

def make_checkerboard_preview(
    rgba_image,
    checker_size=24
):
    """
    Create a checkerboard preview so transparent pixels
    are visibly distinguishable from black pixels.

    This preview is NOT saved to the output file.
    """

    rgba_image = rgba_image.convert(
        "RGBA"
    )

    width, height = rgba_image.size

    background = Image.new(
        "RGB",
        (width, height),
        (230, 230, 230)
    )

    draw = ImageDraw.Draw(
        background
    )

    second_color = (
        190,
        190,
        190
    )

    for y in range(
        0,
        height,
        checker_size
    ):

        for x in range(
            0,
            width,
            checker_size
        ):

            if (
                (
                    x // checker_size
                    +
                    y // checker_size
                )
                %
                2
                ==
                1
            ):

                draw.rectangle(
                    [
                        x,
                        y,
                        min(
                            x + checker_size - 1,
                            width - 1
                        ),
                        min(
                            y + checker_size - 1,
                            height - 1
                        ),
                    ],
                    fill=second_color
                )

    background = background.convert(
        "RGBA"
    )

    preview = Image.alpha_composite(
        background,
        rgba_image
    )

    return preview


# ============================================================
# FILE UPLOAD
# ============================================================

print(
    "Upload PDF, JPG, JPEG, or PNG files."
)

uploaded = files.upload()


# ============================================================
# PROCESS FILES
# ============================================================

output_files = []


for filename, file_data in uploaded.items():

    extension = os.path.splitext(
        filename
    )[1].lower()

    base_name = os.path.splitext(
        os.path.basename(filename)
    )[0]

    print()
    print("=" * 70)
    print(
        f"Processing: {filename}"
    )
    print("=" * 70)

    images = []

    # --------------------------------------------------------
    # PDF
    # --------------------------------------------------------

    if extension == ".pdf":

        images = render_pdf(
            file_data
        )

    # --------------------------------------------------------
    # JPG / JPEG / PNG
    # --------------------------------------------------------

    elif extension in [
        ".jpg",
        ".jpeg",
        ".png"
    ]:

        image = Image.open(
            io.BytesIO(
                file_data
            )
        )

        images = [
            image
        ]

    else:

        print(
            f"Unsupported file type: "
            f"{filename}"
        )

        continue

    # --------------------------------------------------------
    # Process each image / PDF page
    # --------------------------------------------------------

    for index, image in enumerate(
        images
    ):

        print()
        print(
            f"Processing page/image "
            f"{index + 1}..."
        )

        result = process_pil_image(
            image
        )

        if extension == ".pdf":

            output_name = (
                f"{base_name}"
                f"_page_"
                f"{index + 1:03d}"
                f"_transparent.png"
            )

        else:

            output_name = (
                f"{base_name}"
                f"_transparent.png"
            )

        result.save(
            output_name,
            format="PNG"
        )

        output_files.append(
            output_name
        )

        print(
            f"Saved: {output_name}"
        )

        # ----------------------------------------------------
        # Show transparency using checkerboard preview
        # ----------------------------------------------------

        if SHOW_PREVIEW:

            print(
                "Preview:"
                " checkerboard = transparent area"
            )

            preview = make_checkerboard_preview(
                result
            )

            display(
                preview
            )


# ============================================================
# DOWNLOAD RESULTS
# ============================================================

if len(output_files) == 1:

    print()
    print(
        "Finished successfully."
    )

    print(
        f"Output: {output_files[0]}"
    )

    if AUTO_DOWNLOAD:

        files.download(
            output_files[0]
        )


elif len(output_files) > 1:

    zip_name = (
        "transparent_png_results.zip"
    )

    with zipfile.ZipFile(
        zip_name,
        "w",
        compression=zipfile.ZIP_DEFLATED
    ) as zip_file:

        for output_file in output_files:

            zip_file.write(
                output_file,
                arcname=os.path.basename(
                    output_file
                )
            )

    print()
    print(
        f"Finished successfully."
    )

    print(
        f"{len(output_files)} files "
        f"were saved to {zip_name}."
    )

    if AUTO_DOWNLOAD:

        files.download(
            zip_name
        )


else:

    print(
        "No output files were created."
    )