We propose a new neural network architecture for solving single-image analogies - the generation of an entire set of stylistically similar images from just a single input image. Solving this problem requires separating image style from content. Our network is a modified variational autoencoder (VAE) that supports supervised training of single-image analogies and in-network evaluation of outputs with a structured similarity objective that captures pixel covariances. On the challenging task of generating a 62-letter font from a single example letter we produce images with 22.4% lower dissimilarity to the ground truth than state-of-the-art.
@article{arxiv.1603.02003,
title = {From A to Z: Supervised Transfer of Style and Content Using Deep Neural Network Generators},
author = {Paul Upchurch and Noah Snavely and Kavita Bala},
journal= {arXiv preprint arXiv:1603.02003},
year = {2016}
}