Not a member of Pastebin yet?
Sign Up,
it unlocks many cool features!
- import wave, struct, math # To calculate the WAV file content
- import numpy as np # To handle matrices
- from PIL import Image # To open the input image and convert it to grayscale
- import scipy.ndimage # To resample using nearest neighbour
- def imageProcessing(imgArr):
- # Invert filter
- #imgArr = 1 - imgArr
- return imgArr
- '''
- Loads a picture, converts it to greyscale, then to numpy array, normalise it so that the max value is 1
- the min is 0, increase the contrast a bit, remove every pixel which intensity is lower that 0.5,
- then resize the picture using nearest neighbour resampling and outputs the numpy matrix.
- FYI: imgArr[0,0] is the top left corner of the image, cheers matrix indexing
- Returns: the resized image as a high contrast, normalised between 0 and 1, numpy matrix
- '''
- def loadPicture(size, file, verbose=1):
- img = Image.open(file)
- img = img.convert("L")
- #img = img.resize(size) # DO NOT DO THAT OR THE PC WILL CRASH
- imgArr = np.array(img)
- imgArr = np.flip(imgArr, axis=0)
- if verbose:
- print("Image original size: ", imgArr.shape)
- # Scale between 0 and 1
- imgArr -= np.min(imgArr)
- imgArr = imgArr/np.max(imgArr)
- # Processes the image before down sampling
- imgArr = imageProcessing(imgArr)
- if size[0] == 0:
- size = imgArr.shape[0], size[1]
- if size[1] == 0:
- size = size[0], imgArr.shape[1]
- resamplingFactor = size[0]/imgArr.shape[0], size[1]/imgArr.shape[1]
- if resamplingFactor[0] == 0:
- resamplingFactor = 1, resamplingFactor[1]
- if resamplingFactor[1] == 0:
- resamplingFactor = resamplingFactor[0], 1
- # Order : 0=nearestNeighbour, 1:bilinear, 2:cubic etc...
- imgArr = scipy.ndimage.zoom(imgArr, resamplingFactor, order=0)
- if verbose:
- print("Resampling factor", resamplingFactor)
- print("Image resized :", imgArr.shape)
- print("Max intensity: ", np.max(imgArr))
- print("Min intensity: ", np.min(imgArr))
- return imgArr
- def genSoundFromImage(file, output="sound.wav", duration=2.5, sampleRate=44100.0, min_freq=0, max_freq=22000, stepSize=100, substepSize=250, verbose=False):
- wavef = wave.open(output,'w')
- wavef.setnchannels(1) # mono
- wavef.setsampwidth(2)
- wavef.setframerate(sampleRate)
- max_frame = int(duration * sampleRate)
- max_intensity = 32767 # Defined by WAV
- steppingSpectrum = int((max_freq-min_freq)/stepSize)
- imgMat = loadPicture(size=(steppingSpectrum, max_frame), file=file, verbose=verbose)
- imgMat *= max_intensity # To scale it to max WAV audio intensity
- if verbose:
- print("Input: ", file)
- print("Duration (in seconds): ", duration)
- print("Sample rate: ", sampleRate)
- print("Computing each soundframe sum value..")
- for frame in range(max_frame):
- if frame % 120 == 0: # Only print once in a while
- print("Progress: ==> {:.2%}".format(frame/max_frame), end="\r")
- signalValue, count = 0, 0
- for step in range(steppingSpectrum):
- intensity = imgMat[step, frame]
- # nextFreq is less than currentFreq
- currentFreq = (step * stepSize) + min_freq
- nextFreq = ((step+1) * stepSize) + min_freq
- if nextFreq - min_freq > max_freq: # If we're at the end of the spectrum
- nextFreq = max_freq
- for freq in range(currentFreq, nextFreq, substepSize):
- signalValue += intensity*math.cos(freq * 2 * math.pi * float(frame) / float(sampleRate))
- count += 1
- if count == 0: count = 1
- signalValue /= count
- data = struct.pack('<h', int(signalValue))
- wavef.writeframesraw( data )
- wavef.writeframes(''.encode())
- wavef.close()
- print("\nProgress: ==> 100%")
- if verbose:
- print("Output: ", output)
- import sys
- import argparse
- def main(argv):
- parser = argparse.ArgumentParser()
- parser.add_argument("inputImage", help="Input image in any PIL supported format (JPG, PNG (with and without alpha), BMP etc...)")
- parser.add_argument("outputFile", help="path where to output the soundfile in WAV format")
- parser.add_argument("-d", "--duration", help="Duration of the sound to output, in whole seconds, default: 2.5", type=int)
- parser.add_argument("-n", "--minFreq", help="Minimum frequency to use, in Hz, default: 0", type=int)
- parser.add_argument("-x", "--maxFreq", help="Maximum frequency to use, in Hz, default: 22000", type=int)
- parser.add_argument("-s", "--samplerate", help="Sample rate of the sound to output, in Hertz, default: 44100", type=int)
- parser.add_argument("-ss", "--stepsize", help="Each pixel's portion of the spectrum, in Hertz, default: 100", type=int)
- parser.add_argument("-sss", "--substepsize", help="Step between frequencies generated in between two others, in Hertz, default: 250", type=int)
- parser.add_argument("-v", "--verbose", help="Display verbose", action="store_true")
- args = parser.parse_args()
- img = args.inputImage
- output = args.outputFile
- duration = 2.5 if not args.duration else args.duration
- min_freq = 0 if not args.minFreq else args.minFreq
- max_freq = 22000 if not args.maxFreq else args.maxFreq
- sampleRate = 44100 if not args.samplerate else args.samplerate
- stepSize = 100 if not args.stepsize else args.stepsize
- substepSize = 250 if not args.substepsize else args.substepsize
- verbose = args.verbose
- genSoundFromImage(
- file=img,
- output=output,
- duration=duration,
- sampleRate=sampleRate,
- min_freq=min_freq,
- max_freq=max_freq,
- stepSize=stepSize,
- substepSize=substepSize,
- verbose=verbose)
- if __name__ == "__main__":
- main(sys.argv[1:])
Advertisement
Add Comment
Please, Sign In to add comment