GitHub Viewer
"""Perception (Chapter 24)"""
import numpy as np
import scipy.signal
import matplotlib.pyplot as plt
from utils4e import gaussian_kernel_2d
import keras
from keras.datasets import mnist
from keras.models import Sequential
from keras.layers import Dense, Activation, Flatten, InputLayer
from keras.layers import Conv2D, MaxPooling2D
import cv2
# ____________________________________________________
# 24.3 Early Image Processing Operators
# 24.3.1 Edge Detection
def array_normalization(array, range_min, range_max):
"""normalize an array in the range of (range_min, range_max)"""
if not isinstance(array, np.ndarray):
array = np.asarray(array)
array = array - np.min(array)
array = array * (range_max - range_min) / np.max(array) + range_min
return array
def gradient_edge_detector(image):
"""
Image edge detection by calculating gradients in the image
:param image: numpy ndarray or an iterable object
:return: numpy ndarray, representing a gray scale image
"""
if not isinstance(image, np.ndarray):
image = np.asarray(image)
# gradient filters of x and y direction edges
x_filter, y_filter = np.array([[1, -1]]), np.array([[1], [-1]])
# convolution between filter and image to get edges
y_edges = scipy.signal.convolve2d(image, x_filter, 'same')
x_edges = scipy.signal.convolve2d(image, y_filter, 'same')
edges = array_normalization(x_edges+y_edges, 0, 255)
return edges
def gaussian_derivative_edge_detector(image):
"""Image edge detector using derivative of gaussian kernels"""
if not isinstance(image, np.ndarray):
image = np.asarray(image)
gaussian_filter = gaussian_kernel_2d()
# init derivative of gaussian filters
x_filter = scipy.signal.convolve2d(gaussian_filter, np.asarray([[1, -1]]), 'same')
y_filter = scipy.signal.convolve2d(gaussian_filter, np.asarray([[1], [-1]]), 'same')
# extract edges using convolution
y_edges = scipy.signal.convolve2d(image, x_filter, 'same')
x_edges = scipy.signal.convolve2d(image, y_filter, 'same')
edges = array_normalization(x_edges+y_edges, 0, 255)
return edges
def laplacian_edge_detector(image):
"""Extract image edge with laplacian filter"""
if not isinstance(image, np.ndarray):
image = np.asarray(image)
# init laplacian filter
laplacian_kernel = np.asarray([[0, -1, 0], [-1, 4, -1], [0, -1, 0]])
# extract edges with convolution
edges = scipy.signal.convolve2d(image, laplacian_kernel, 'same')
edges = array_normalization(edges, 0, 255)
return edges
def show_edges(edges):
""" helper function to show edges picture"""
plt.imshow(edges, cmap='gray', vmin=0, vmax=255)
plt.axis('off')
plt.show()
# __________________________________________________
# 24.3.3 Optical flow
def sum_squared_difference(pic1, pic2):
"""ssd of two frames"""
pic1 = np.asarray(pic1)
pic2 = np.asarray(pic2)
assert pic1.shape == pic2.shape
min_ssd = float('inf')
min_dxy = (float('inf'), float('inf'))
# consider picture shift from -30 to 30
for Dx in range(-30, 31):
for Dy in range(-30, 31):
# shift the image
shifted_pic = np.roll(pic2, Dx, axis=0)
shifted_pic = np.roll(shifted_pic, Dy, axis=1)
# calculate the difference
diff = np.sum((pic1 - shifted_pic) ** 2)
if diff < min_ssd:
min_dxy = (Dx, Dy)
min_ssd = diff
return min_dxy, min_ssd
# ____________________________________________________
# segmentation
def gen_gray_scale_picture(size, level=3):
"""
Generate a picture with different gray scale levels
:param size: size of generated picture
:param level: the number of level of gray scales in the picture,
range (0, 255) are equally divided by number of levels
:return image in numpy ndarray type
"""
assert level > 0
# init an empty image
image = np.zeros((size, size))
if level == 1:
return image
# draw a square on the left upper corner of the image
for x in range(size):
for y in range(size):
image[x,y] += (250//(level-1)) * (max(x, y)*level//size)
return image
gray_scale_image = gen_gray_scale_picture(3)
def probability_contour_detection(image, discs, threshold=0):
"""
detect edges/contours by applying a set of discs to an image
:param image: an image in type of numpy ndarray
:param discs: a set of discs/filters to apply to pixels of image
:param threshold: threshold to tell whether the pixel at (x, y) is on an edge
:return image showing edges in numpy ndarray type
"""
# init an empty output image
res = np.zeros(image.shape)
step = discs[0].shape[0]
for x_i in range(0, image.shape[0]-step+1,1):
for y_i in range(0, image.shape[1]-step+1, 1):
diff = []
# apply each pair of discs and calculate the difference
for d in range(0, len(discs),2):
disc1, disc2 = discs[d], discs[d+1]
# crop the region of interest
region = image[x_i: x_i+step, y_i: y_i+step]
diff.append(np.sum(np.multiply(region, disc1)) - np.sum(np.multiply(region, disc2)))
if max(diff) > threshold:
# change color of the center of region
res[x_i + step//2, y_i + step//2] = 255
return res
def group_contour_detection(image, cluster_num=2):
"""
detecting contours in an image with k-means clustering
:param image: an image in numpy ndarray type
:param cluster_num: number of clusters in k-means
"""
img = image
Z = np.float32(img)
criteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, 10, 1.0)
K = cluster_num
# use kmeans in opencv-python
ret, label, center = cv2.kmeans(Z, K, None, criteria, 10, cv2.KMEANS_RANDOM_CENTERS)
center = np.uint8(center)
res = center[label.flatten()]
res2 = res.reshape((img.shape))
# show the image
# cv2.imshow('res2', res2)
# cv2.waitKey(0)
# cv2.destroyAllWindows()
return res2
def image_to_graph(image):
"""
convert an image to an graph in adjacent matrix form
"""
graph_dict = {}
for x in range(image.shape[0]):
for y in range(image.shape[1]):
graph_dict[(x, y)] = [(x+1, y) if x+1 < image.shape[0] else None, (x, y+1) if y+1 < image.shape[1] else None]
return graph_dict
def generate_edge_weight(image, v1, v2):
"""
find edge weight between two vertices in an image
:param image: image in numpy ndarray type
:param v1, v2: verticles in the image in form of (x index, y index)
"""
diff = abs(image[v1[0], v1[1]] - image[v2[0], v2[1]])
return 255-diff
class Graph:
"""graph in adjacent matrix to represent an image"""
def __init__(self, image):
"""image: ndarray"""
self.graph = image_to_graph(image)
# number of columns and rows
self.ROW = len(self.graph)
self.COL = 2
self.image = image
# dictionary to save the maximum flow of each edge
self.flow = {}
# initialize the flow
for s in self.graph:
self.flow[s] = {}
for t in self.graph[s]:
if t:
self.flow[s][t] = generate_edge_weight(image, s, t)
def bfs(self, s, t, parent):
"""breadth first search to tell whether there is an edge between source and sink
parent: a list to save the path between s and t"""
# queue to save the current searching frontier
queue = [s]
visited = []
while queue:
u = queue.pop(0)
for node in self.graph[u]:
# only select edge with positive flow
if node not in visited and node and self.flow[u][node]>0:
queue.append(node)
visited.append(node)
parent.append((u, node))
return True if t in visited else False
def min_cut(self, source, sink):
"""find the minimum cut of the graph between source and sink"""
parent = []
max_flow = 0
while self.bfs(source, sink, parent):
path_flow = float('inf')
# find the minimum flow of s-t path
for s, t in parent:
path_flow = min(path_flow, self.flow[s][t])
max_flow += path_flow
# update all edges between source and sink
for s in self.flow:
for t in self.flow[s]:
if t[0]