Speech Based Object Manipulation in Unity using OpenAI's Whisper Model
C#
0
10 commits
updated Oct 4, 2025
A Unity-based 3D application that enables voice-controlled object manipulation using Python Speechrecognition model for natural language processing.
# 1. Clone the repository
git clone https://github.com/haotianli24/object-manipulation.git
cd object-manipulation
# 2. Install Python dependencies
pip install openai-whisper
pip install pyaudio numpy
# 3. Open in Unity
# Open Unity Hub and add the project folder
Configure Audio Input
Whisper Model Setup
// Example voice commands the system recognizes:
"Move the cube to the right"
"Rotate the sphere"
"Scale the object up"
"Change color to red"
"Delete the cylinder"
"Create a new box"
public class VoiceController : MonoBehaviour
{
[SerializeField] private AudioSource audioSource;
[SerializeField] private GameObject[] manipulableObjects;
private void Update()
{
if (Input.GetKeyDown(KeyCode.Space))
{
StartRecording();
}
}
private void ProcessVoiceCommand(string command)
{
// Parse command and execute corresponding action
if (command.Contains("move") && command.Contains("cube"))
{
MoveCube(ExtractDirection(command));
}
else if (command.Contains("rotate"))
{
RotateObject(ExtractObject(command));
}
}
}
public class ObjectManipulator : MonoBehaviour
{
public void MoveObject(Vector3 direction)
{
transform.Translate(direction * moveSpeed * Time.deltaTime);
}
public void RotateObject(Vector3 axis)
{
transform.Rotate(axis * rotationSpeed * Time.deltaTime);
}
public void ScaleObject(float scaleFactor)
{
transform.localScale *= scaleFactor;
}
}
object-manipulation/
├── Assets/
│ ├── Scripts/
│ │ ├── VoiceController.cs # Speech recognition handler
│ │ ├── ObjectManipulator.cs # Object control logic
│ │ └── CommandParser.cs # NLP command parsing
│ ├── Scenes/
│ │ └── MainScene.unity # Main interaction scene
│ ├── Prefabs/
│ │ └── InteractableObjects/ # 3D object prefabs
│ └── StreamingAssets/
│ └── WhisperModels/ # AI model files
├── Python/
│ ├── whisper_server.py # Python speech processing
│ └── requirements.txt # Python dependencies
└── README.md
# Python backend for speech recognition
import whisper
import pyaudio
class SpeechProcessor:
def __init__(self, model_size="base"):
self.model = whisper.load_model(model_size)
def transcribe_audio(self, audio_file):
result = self.model.transcribe(audio_file)
return result["text"]
def process_command(self, text):
# Parse natural language commands
return self.parse_intent(text)
public class WhisperIntegration : MonoBehaviour
{
private Process pythonProcess;
void Start()
{
StartPythonBackend();
}
private void StartPythonBackend()
{
pythonProcess = new Process();
pythonProcess.StartInfo.FileName = "python";
pythonProcess.StartInfo.Arguments = "Python/whisper_server.py";
pythonProcess.Start();
}
}
| Command Type | Example | Action |
|---|---|---|
| Movement | "Move cube left" | Translate object |
| Rotation | "Rotate sphere" | Apply rotation |
| Scaling | "Make it bigger" | Scale transform |
| Color | "Turn it red" | Change material |
| Creation | "Create a cylinder" | Instantiate prefab |
| Deletion | "Delete the box" | Destroy object |
# Start the application
cd object-manipulation
python Python/whisper_server.py # Start Python backend
# Open Unity and press Play
# Test voice recognition
python Python/test_microphone.py
# Install additional models
python -c "import whisper; whisper.load_model('medium')"
// Movement commands
"Move [object] [direction]"
"Translate the [object] to [position]"
// Rotation commands
"Rotate [object] [axis/direction]"
"Turn the [object] around"
// Scaling commands
"Scale [object] [up/down/factor]"
"Make the [object] [bigger/smaller]"
// Material commands
"Change [object] color to [color]"
"Make [object] [transparent/opaque]"
Unity Package Dependencies:
Python Dependencies:
openai-whisper>=20230314
pyaudio>=0.2.11
numpy>=1.21.0
scipy>=1.7.0
Built for intuitive 3D interaction through natural speech
C#
99.0%
Speech Based Object Manipulation in Unity using OpenAI's Whisper Model
C#
0
10 commits
updated Oct 4, 2025
A Unity-based 3D application that enables voice-controlled object manipulation using Python Speechrecognition model for natural language processing.
# 1. Clone the repository
git clone https://github.com/haotianli24/object-manipulation.git
cd object-manipulation
# 2. Install Python dependencies
pip install openai-whisper
pip install pyaudio numpy
# 3. Open in Unity
# Open Unity Hub and add the project folder
Configure Audio Input
Whisper Model Setup
// Example voice commands the system recognizes:
"Move the cube to the right"
"Rotate the sphere"
"Scale the object up"
"Change color to red"
"Delete the cylinder"
"Create a new box"
public class VoiceController : MonoBehaviour
{
[SerializeField] private AudioSource audioSource;
[SerializeField] private GameObject[] manipulableObjects;
private void Update()
{
if (Input.GetKeyDown(KeyCode.Space))
{
StartRecording();
}
}
private void ProcessVoiceCommand(string command)
{
// Parse command and execute corresponding action
if (command.Contains("move") && command.Contains("cube"))
{
MoveCube(ExtractDirection(command));
}
else if (command.Contains("rotate"))
{
RotateObject(ExtractObject(command));
}
}
}
public class ObjectManipulator : MonoBehaviour
{
public void MoveObject(Vector3 direction)
{
transform.Translate(direction * moveSpeed * Time.deltaTime);
}
public void RotateObject(Vector3 axis)
{
transform.Rotate(axis * rotationSpeed * Time.deltaTime);
}
public void ScaleObject(float scaleFactor)
{
transform.localScale *= scaleFactor;
}
}
object-manipulation/
├── Assets/
│ ├── Scripts/
│ │ ├── VoiceController.cs # Speech recognition handler
│ │ ├── ObjectManipulator.cs # Object control logic
│ │ └── CommandParser.cs # NLP command parsing
│ ├── Scenes/
│ │ └── MainScene.unity # Main interaction scene
│ ├── Prefabs/
│ │ └── InteractableObjects/ # 3D object prefabs
│ └── StreamingAssets/
│ └── WhisperModels/ # AI model files
├── Python/
│ ├── whisper_server.py # Python speech processing
│ └── requirements.txt # Python dependencies
└── README.md
# Python backend for speech recognition
import whisper
import pyaudio
class SpeechProcessor:
def __init__(self, model_size="base"):
self.model = whisper.load_model(model_size)
def transcribe_audio(self, audio_file):
result = self.model.transcribe(audio_file)
return result["text"]
def process_command(self, text):
# Parse natural language commands
return self.parse_intent(text)
public class WhisperIntegration : MonoBehaviour
{
private Process pythonProcess;
void Start()
{
StartPythonBackend();
}
private void StartPythonBackend()
{
pythonProcess = new Process();
pythonProcess.StartInfo.FileName = "python";
pythonProcess.StartInfo.Arguments = "Python/whisper_server.py";
pythonProcess.Start();
}
}
| Command Type | Example | Action |
|---|---|---|
| Movement | "Move cube left" | Translate object |
| Rotation | "Rotate sphere" | Apply rotation |
| Scaling | "Make it bigger" | Scale transform |
| Color | "Turn it red" | Change material |
| Creation | "Create a cylinder" | Instantiate prefab |
| Deletion | "Delete the box" | Destroy object |
# Start the application
cd object-manipulation
python Python/whisper_server.py # Start Python backend
# Open Unity and press Play
# Test voice recognition
python Python/test_microphone.py
# Install additional models
python -c "import whisper; whisper.load_model('medium')"
// Movement commands
"Move [object] [direction]"
"Translate the [object] to [position]"
// Rotation commands
"Rotate [object] [axis/direction]"
"Turn the [object] around"
// Scaling commands
"Scale [object] [up/down/factor]"
"Make the [object] [bigger/smaller]"
// Material commands
"Change [object] color to [color]"
"Make [object] [transparent/opaque]"
Unity Package Dependencies:
Python Dependencies:
openai-whisper>=20230314
pyaudio>=0.2.11
numpy>=1.21.0
scipy>=1.7.0
Built for intuitive 3D interaction through natural speech
C#
99.0%