Initial commit
This commit is contained in:
@@ -0,0 +1,74 @@
|
||||
import torch
|
||||
|
||||
|
||||
def get_focus_rate(attn, src_padding_mask=None, tgt_padding_mask=None):
|
||||
'''
|
||||
attn: bs x L_t x L_s
|
||||
'''
|
||||
if src_padding_mask is not None:
|
||||
attn = attn * (1 - src_padding_mask.float())[:, None, :]
|
||||
|
||||
if tgt_padding_mask is not None:
|
||||
attn = attn * (1 - tgt_padding_mask.float())[:, :, None]
|
||||
|
||||
focus_rate = attn.max(-1).values.sum(-1)
|
||||
focus_rate = focus_rate / attn.sum(-1).sum(-1)
|
||||
return focus_rate
|
||||
|
||||
|
||||
def get_phone_coverage_rate(attn, src_padding_mask=None, src_seg_mask=None, tgt_padding_mask=None):
|
||||
'''
|
||||
attn: bs x L_t x L_s
|
||||
'''
|
||||
src_mask = attn.new(attn.size(0), attn.size(-1)).bool().fill_(False)
|
||||
if src_padding_mask is not None:
|
||||
src_mask |= src_padding_mask
|
||||
if src_seg_mask is not None:
|
||||
src_mask |= src_seg_mask
|
||||
|
||||
attn = attn * (1 - src_mask.float())[:, None, :]
|
||||
if tgt_padding_mask is not None:
|
||||
attn = attn * (1 - tgt_padding_mask.float())[:, :, None]
|
||||
|
||||
phone_coverage_rate = attn.max(1).values.sum(-1)
|
||||
# phone_coverage_rate = phone_coverage_rate / attn.sum(-1).sum(-1)
|
||||
phone_coverage_rate = phone_coverage_rate / (1 - src_mask.float()).sum(-1)
|
||||
return phone_coverage_rate
|
||||
|
||||
|
||||
def get_diagonal_focus_rate(attn, attn_ks, target_len, src_padding_mask=None, tgt_padding_mask=None,
|
||||
band_mask_factor=5, band_width=50):
|
||||
'''
|
||||
attn: bx x L_t x L_s
|
||||
attn_ks: shape: tensor with shape [batch_size], input_lens/output_lens
|
||||
|
||||
diagonal: y=k*x (k=attn_ks, x:output, y:input)
|
||||
1 0 0
|
||||
0 1 0
|
||||
0 0 1
|
||||
y>=k*(x-width) and y<=k*(x+width):1
|
||||
else:0
|
||||
'''
|
||||
# width = min(target_len/band_mask_factor, 50)
|
||||
width1 = target_len / band_mask_factor
|
||||
width2 = target_len.new(target_len.size()).fill_(band_width)
|
||||
width = torch.where(width1 < width2, width1, width2).float()
|
||||
base = torch.ones(attn.size()).to(attn.device)
|
||||
zero = torch.zeros(attn.size()).to(attn.device)
|
||||
x = torch.arange(0, attn.size(1)).to(attn.device)[None, :, None].float() * base
|
||||
y = torch.arange(0, attn.size(2)).to(attn.device)[None, None, :].float() * base
|
||||
cond = (y - attn_ks[:, None, None] * x)
|
||||
cond1 = cond + attn_ks[:, None, None] * width[:, None, None]
|
||||
cond2 = cond - attn_ks[:, None, None] * width[:, None, None]
|
||||
mask1 = torch.where(cond1 < 0, zero, base)
|
||||
mask2 = torch.where(cond2 > 0, zero, base)
|
||||
mask = mask1 * mask2
|
||||
|
||||
if src_padding_mask is not None:
|
||||
attn = attn * (1 - src_padding_mask.float())[:, None, :]
|
||||
if tgt_padding_mask is not None:
|
||||
attn = attn * (1 - tgt_padding_mask.float())[:, :, None]
|
||||
|
||||
diagonal_attn = attn * mask
|
||||
diagonal_focus_rate = diagonal_attn.sum(-1).sum(-1) / attn.sum(-1).sum(-1)
|
||||
return diagonal_focus_rate, mask
|
||||
@@ -0,0 +1,160 @@
|
||||
from numpy import array, zeros, full, argmin, inf, ndim
|
||||
from scipy.spatial.distance import cdist
|
||||
from math import isinf
|
||||
|
||||
|
||||
def dtw(x, y, dist, warp=1, w=inf, s=1.0):
|
||||
"""
|
||||
Computes Dynamic Time Warping (DTW) of two sequences.
|
||||
|
||||
:param array x: N1*M array
|
||||
:param array y: N2*M array
|
||||
:param func dist: distance used as cost measure
|
||||
:param int warp: how many shifts are computed.
|
||||
:param int w: window size limiting the maximal distance between indices of matched entries |i,j|.
|
||||
:param float s: weight applied on off-diagonal moves of the path. As s gets larger, the warping path is increasingly biased towards the diagonal
|
||||
Returns the minimum distance, the cost matrix, the accumulated cost matrix, and the wrap path.
|
||||
"""
|
||||
assert len(x)
|
||||
assert len(y)
|
||||
assert isinf(w) or (w >= abs(len(x) - len(y)))
|
||||
assert s > 0
|
||||
r, c = len(x), len(y)
|
||||
if not isinf(w):
|
||||
D0 = full((r + 1, c + 1), inf)
|
||||
for i in range(1, r + 1):
|
||||
D0[i, max(1, i - w):min(c + 1, i + w + 1)] = 0
|
||||
D0[0, 0] = 0
|
||||
else:
|
||||
D0 = zeros((r + 1, c + 1))
|
||||
D0[0, 1:] = inf
|
||||
D0[1:, 0] = inf
|
||||
D1 = D0[1:, 1:] # view
|
||||
for i in range(r):
|
||||
for j in range(c):
|
||||
if (isinf(w) or (max(0, i - w) <= j <= min(c, i + w))):
|
||||
D1[i, j] = dist(x[i], y[j])
|
||||
C = D1.copy()
|
||||
jrange = range(c)
|
||||
for i in range(r):
|
||||
if not isinf(w):
|
||||
jrange = range(max(0, i - w), min(c, i + w + 1))
|
||||
for j in jrange:
|
||||
min_list = [D0[i, j]]
|
||||
for k in range(1, warp + 1):
|
||||
i_k = min(i + k, r)
|
||||
j_k = min(j + k, c)
|
||||
min_list += [D0[i_k, j] * s, D0[i, j_k] * s]
|
||||
D1[i, j] += min(min_list)
|
||||
if len(x) == 1:
|
||||
path = zeros(len(y)), range(len(y))
|
||||
elif len(y) == 1:
|
||||
path = range(len(x)), zeros(len(x))
|
||||
else:
|
||||
path = _traceback(D0)
|
||||
return D1[-1, -1], C, D1, path
|
||||
|
||||
|
||||
def accelerated_dtw(x, y, dist, warp=1):
|
||||
"""
|
||||
Computes Dynamic Time Warping (DTW) of two sequences in a faster way.
|
||||
Instead of iterating through each element and calculating each distance,
|
||||
this uses the cdist function from scipy (https://docs.scipy.org/doc/scipy/reference/generated/scipy.spatial.distance.cdist.html)
|
||||
|
||||
:param array x: N1*M array
|
||||
:param array y: N2*M array
|
||||
:param string or func dist: distance parameter for cdist. When string is given, cdist uses optimized functions for the distance metrics.
|
||||
If a string is passed, the distance function can be 'braycurtis', 'canberra', 'chebyshev', 'cityblock', 'correlation', 'cosine', 'dice', 'euclidean', 'hamming', 'jaccard', 'kulsinski', 'mahalanobis', 'matching', 'minkowski', 'rogerstanimoto', 'russellrao', 'seuclidean', 'sokalmichener', 'sokalsneath', 'sqeuclidean', 'wminkowski', 'yule'.
|
||||
:param int warp: how many shifts are computed.
|
||||
Returns the minimum distance, the cost matrix, the accumulated cost matrix, and the wrap path.
|
||||
"""
|
||||
assert len(x)
|
||||
assert len(y)
|
||||
if ndim(x) == 1:
|
||||
x = x.reshape(-1, 1)
|
||||
if ndim(y) == 1:
|
||||
y = y.reshape(-1, 1)
|
||||
r, c = len(x), len(y)
|
||||
D0 = zeros((r + 1, c + 1))
|
||||
D0[0, 1:] = inf
|
||||
D0[1:, 0] = inf
|
||||
D1 = D0[1:, 1:]
|
||||
D0[1:, 1:] = cdist(x, y, dist)
|
||||
C = D1.copy()
|
||||
for i in range(r):
|
||||
for j in range(c):
|
||||
min_list = [D0[i, j]]
|
||||
for k in range(1, warp + 1):
|
||||
min_list += [D0[min(i + k, r), j],
|
||||
D0[i, min(j + k, c)]]
|
||||
D1[i, j] += min(min_list)
|
||||
if len(x) == 1:
|
||||
path = zeros(len(y)), range(len(y))
|
||||
elif len(y) == 1:
|
||||
path = range(len(x)), zeros(len(x))
|
||||
else:
|
||||
path = _traceback(D0)
|
||||
return D1[-1, -1], C, D1, path
|
||||
|
||||
|
||||
def _traceback(D):
|
||||
i, j = array(D.shape) - 2
|
||||
p, q = [i], [j]
|
||||
while (i > 0) or (j > 0):
|
||||
tb = argmin((D[i, j], D[i, j + 1], D[i + 1, j]))
|
||||
if tb == 0:
|
||||
i -= 1
|
||||
j -= 1
|
||||
elif tb == 1:
|
||||
i -= 1
|
||||
else: # (tb == 2):
|
||||
j -= 1
|
||||
p.insert(0, i)
|
||||
q.insert(0, j)
|
||||
return array(p), array(q)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
w = inf
|
||||
s = 1.0
|
||||
if 1: # 1-D numeric
|
||||
from sklearn.metrics.pairwise import manhattan_distances
|
||||
|
||||
x = [0, 0, 1, 1, 2, 4, 2, 1, 2, 0]
|
||||
y = [1, 1, 1, 2, 2, 2, 2, 3, 2, 0]
|
||||
dist_fun = manhattan_distances
|
||||
w = 1
|
||||
# s = 1.2
|
||||
elif 0: # 2-D numeric
|
||||
from sklearn.metrics.pairwise import euclidean_distances
|
||||
|
||||
x = [[0, 0], [0, 1], [1, 1], [1, 2], [2, 2], [4, 3], [2, 3], [1, 1], [2, 2], [0, 1]]
|
||||
y = [[1, 0], [1, 1], [1, 1], [2, 1], [4, 3], [4, 3], [2, 3], [3, 1], [1, 2], [1, 0]]
|
||||
dist_fun = euclidean_distances
|
||||
else: # 1-D list of strings
|
||||
from nltk.metrics.distance import edit_distance
|
||||
|
||||
# x = ['we', 'shelled', 'clams', 'for', 'the', 'chowder']
|
||||
# y = ['class', 'too']
|
||||
x = ['i', 'soon', 'found', 'myself', 'muttering', 'to', 'the', 'walls']
|
||||
y = ['see', 'drown', 'himself']
|
||||
# x = 'we talked about the situation'.split()
|
||||
# y = 'we talked about the situation'.split()
|
||||
dist_fun = edit_distance
|
||||
dist, cost, acc, path = dtw(x, y, dist_fun, w=w, s=s)
|
||||
|
||||
# Vizualize
|
||||
from matplotlib import pyplot as plt
|
||||
|
||||
plt.imshow(cost.T, origin='lower', cmap=plt.cm.Reds, interpolation='nearest')
|
||||
plt.plot(path[0], path[1], '-o') # relation
|
||||
plt.xticks(range(len(x)), x)
|
||||
plt.yticks(range(len(y)), y)
|
||||
plt.xlabel('x')
|
||||
plt.ylabel('y')
|
||||
plt.axis('tight')
|
||||
if isinf(w):
|
||||
plt.title('Minimum distance: {}, slope weight: {}'.format(dist, s))
|
||||
else:
|
||||
plt.title('Minimum distance: {}, window widht: {}, slope weight: {}'.format(dist, w, s))
|
||||
plt.show()
|
||||
@@ -0,0 +1,4 @@
|
||||
import scipy.ndimage
|
||||
|
||||
def laplace_var(x):
|
||||
return scipy.ndimage.laplace(x).var()
|
||||
@@ -0,0 +1,102 @@
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from numba import jit
|
||||
|
||||
import torch
|
||||
|
||||
|
||||
@jit
|
||||
def time_warp(costs):
|
||||
dtw = np.zeros_like(costs)
|
||||
dtw[0, 1:] = np.inf
|
||||
dtw[1:, 0] = np.inf
|
||||
eps = 1e-4
|
||||
for i in range(1, costs.shape[0]):
|
||||
for j in range(1, costs.shape[1]):
|
||||
dtw[i, j] = costs[i, j] + min(dtw[i - 1, j], dtw[i, j - 1], dtw[i - 1, j - 1])
|
||||
return dtw
|
||||
|
||||
|
||||
def align_from_distances(distance_matrix, debug=False, return_mindist=False):
|
||||
# for each position in spectrum 1, returns best match position in spectrum2
|
||||
# using monotonic alignment
|
||||
dtw = time_warp(distance_matrix)
|
||||
|
||||
i = distance_matrix.shape[0] - 1
|
||||
j = distance_matrix.shape[1] - 1
|
||||
results = [0] * distance_matrix.shape[0]
|
||||
while i > 0 and j > 0:
|
||||
results[i] = j
|
||||
i, j = min([(i - 1, j), (i, j - 1), (i - 1, j - 1)], key=lambda x: dtw[x[0], x[1]])
|
||||
|
||||
if debug:
|
||||
visual = np.zeros_like(dtw)
|
||||
visual[range(len(results)), results] = 1
|
||||
plt.matshow(visual)
|
||||
plt.show()
|
||||
if return_mindist:
|
||||
return results, dtw[-1, -1]
|
||||
return results
|
||||
|
||||
|
||||
def get_local_context(input_f, max_window=32, scale_factor=1.):
|
||||
# input_f: [S, 1], support numpy array or torch tensor
|
||||
# return hist: [S, max_window * 2], list of list
|
||||
T = input_f.shape[0]
|
||||
# max_window = int(max_window * scale_factor)
|
||||
derivative = [[0 for _ in range(max_window * 2)] for _ in range(T)]
|
||||
|
||||
for t in range(T): # travel the time series
|
||||
for feat_idx in range(-max_window, max_window):
|
||||
if t + feat_idx < 0 or t + feat_idx >= T:
|
||||
value = 0
|
||||
else:
|
||||
value = input_f[t + feat_idx]
|
||||
derivative[t][feat_idx + max_window] = value
|
||||
return derivative
|
||||
|
||||
|
||||
def cal_localnorm_dist(src, tgt, src_len, tgt_len):
|
||||
local_src = torch.tensor(get_local_context(src))
|
||||
local_tgt = torch.tensor(get_local_context(tgt, scale_factor=tgt_len / src_len))
|
||||
|
||||
local_norm_src = (local_src - local_src.mean(-1).unsqueeze(-1)) # / local_src.std(-1).unsqueeze(-1) # [T1, 32]
|
||||
local_norm_tgt = (local_tgt - local_tgt.mean(-1).unsqueeze(-1)) # / local_tgt.std(-1).unsqueeze(-1) # [T2, 32]
|
||||
|
||||
dists = torch.cdist(local_norm_src[None, :, :], local_norm_tgt[None, :, :]) # [1, T1, T2]
|
||||
return dists
|
||||
|
||||
|
||||
## here is API for one sample
|
||||
def LoNDTWDistance(src, tgt):
|
||||
# src: [S]
|
||||
# tgt: [T]
|
||||
dists = cal_localnorm_dist(src, tgt, src.shape[0], tgt.shape[0]) # [1, S, T]
|
||||
costs = dists.squeeze(0) # [S, T]
|
||||
alignment, min_distance = align_from_distances(costs.T.cpu().detach().numpy(), return_mindist=True) # [T]
|
||||
return alignment, min_distance
|
||||
|
||||
# if __name__ == '__main__':
|
||||
# # utils from ns
|
||||
# from utils.pitch_utils import denorm_f0
|
||||
# from tasks.singing.fsinging import FastSingingDataset
|
||||
# from utils.hparams import hparams, set_hparams
|
||||
#
|
||||
# set_hparams()
|
||||
#
|
||||
# train_ds = FastSingingDataset('test')
|
||||
#
|
||||
# # Test One sample case
|
||||
# sample = train_ds[0]
|
||||
# amateur_f0 = sample['f0']
|
||||
# prof_f0 = sample['prof_f0']
|
||||
#
|
||||
# amateur_uv = sample['uv']
|
||||
# amateur_padding = sample['mel2ph'] == 0
|
||||
# prof_uv = sample['prof_uv']
|
||||
# prof_padding = sample['prof_mel2ph'] == 0
|
||||
# amateur_f0_denorm = denorm_f0(amateur_f0, amateur_uv, hparams, pitch_padding=amateur_padding)
|
||||
# prof_f0_denorm = denorm_f0(prof_f0, prof_uv, hparams, pitch_padding=prof_padding)
|
||||
# alignment, min_distance = LoNDTWDistance(amateur_f0_denorm, prof_f0_denorm)
|
||||
# print(min_distance)
|
||||
# python utils/pitch_distance.py --config egs/datasets/audio/molar/svc_ppg.yaml
|
||||
@@ -0,0 +1,84 @@
|
||||
"""
|
||||
Adapted from https://github.com/Po-Hsun-Su/pytorch-ssim
|
||||
"""
|
||||
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
from torch.autograd import Variable
|
||||
import numpy as np
|
||||
from math import exp
|
||||
|
||||
|
||||
def gaussian(window_size, sigma):
|
||||
gauss = torch.Tensor([exp(-(x - window_size // 2) ** 2 / float(2 * sigma ** 2)) for x in range(window_size)])
|
||||
return gauss / gauss.sum()
|
||||
|
||||
|
||||
def create_window(window_size, channel):
|
||||
_1D_window = gaussian(window_size, 1.5).unsqueeze(1)
|
||||
_2D_window = _1D_window.mm(_1D_window.t()).float().unsqueeze(0).unsqueeze(0)
|
||||
window = Variable(_2D_window.expand(channel, 1, window_size, window_size).contiguous())
|
||||
return window
|
||||
|
||||
|
||||
def _ssim(img1, img2, window, window_size, channel, size_average=True):
|
||||
mu1 = F.conv2d(img1, window, padding=window_size // 2, groups=channel)
|
||||
mu2 = F.conv2d(img2, window, padding=window_size // 2, groups=channel)
|
||||
|
||||
mu1_sq = mu1.pow(2)
|
||||
mu2_sq = mu2.pow(2)
|
||||
mu1_mu2 = mu1 * mu2
|
||||
|
||||
sigma1_sq = F.conv2d(img1 * img1, window, padding=window_size // 2, groups=channel) - mu1_sq
|
||||
sigma2_sq = F.conv2d(img2 * img2, window, padding=window_size // 2, groups=channel) - mu2_sq
|
||||
sigma12 = F.conv2d(img1 * img2, window, padding=window_size // 2, groups=channel) - mu1_mu2
|
||||
|
||||
C1 = 0.01 ** 2
|
||||
C2 = 0.03 ** 2
|
||||
|
||||
ssim_map = ((2 * mu1_mu2 + C1) * (2 * sigma12 + C2)) / ((mu1_sq + mu2_sq + C1) * (sigma1_sq + sigma2_sq + C2))
|
||||
|
||||
if size_average:
|
||||
return ssim_map.mean()
|
||||
else:
|
||||
return ssim_map.mean(1)
|
||||
|
||||
|
||||
class SSIM(torch.nn.Module):
|
||||
def __init__(self, window_size=11, size_average=True):
|
||||
super(SSIM, self).__init__()
|
||||
self.window_size = window_size
|
||||
self.size_average = size_average
|
||||
self.channel = 1
|
||||
self.window = create_window(window_size, self.channel)
|
||||
|
||||
def forward(self, img1, img2):
|
||||
(_, channel, _, _) = img1.size()
|
||||
|
||||
if channel == self.channel and self.window.data.type() == img1.data.type():
|
||||
window = self.window
|
||||
else:
|
||||
window = create_window(self.window_size, channel)
|
||||
|
||||
if img1.is_cuda:
|
||||
window = window.cuda(img1.get_device())
|
||||
window = window.type_as(img1)
|
||||
|
||||
self.window = window
|
||||
self.channel = channel
|
||||
|
||||
return _ssim(img1, img2, window, self.window_size, channel, self.size_average)
|
||||
|
||||
|
||||
window = None
|
||||
|
||||
|
||||
def ssim(img1, img2, window_size=11, size_average=True):
|
||||
(_, channel, _, _) = img1.size()
|
||||
global window
|
||||
if window is None:
|
||||
window = create_window(window_size, channel)
|
||||
if img1.is_cuda:
|
||||
window = window.cuda(img1.get_device())
|
||||
window = window.type_as(img1)
|
||||
return _ssim(img1, img2, window, window_size, channel, size_average)
|
||||
Reference in New Issue
Block a user