host: cross-platform audio decode (Android has sound)

Replace the macOS-only AudioToolbox decoder with mt_host::decode_audio: a
RIFF/WAVE PCM parser plus vendored dr_mp3, resampled/remixed to 44.1 kHz
stereo float with SDL_ConvertAudioSamples. macOS and Android now share one
path; the AudioToolbox link is gone.

BGM is no longer cached in clips_ (a 5-minute track is ~100 MB of float):
only the playing track is kept, and a new one decodes on a worker thread
(the pack read stays on the main thread) so map changes do not stall.

host.audio_decode test: synthetic wav layouts + all 1578 pack .wav, 9 pack
.mp3 and 25 Client/BGM .mp3 decode (7060 s of audio in ~3.6 s on M2).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
shenlei
2026-09-29 21:22:11 +09:00
co-authored by Claude Opus 5.5
parent 75d2dd8545
commit eab5938072
11 changed files with 5889 additions and 107 deletions
+13 -4
View File
@@ -14,6 +14,7 @@ add_custom_target(mt_native_shaders DEPENDS "${MT_NATIVE_VERT_SPV}" "${MT_NATIVE
set(MT_NATIVE_RENDER_SOURCES
main.cpp
audio_decode.cpp
frame_policy.cpp
host_util.cpp
input_keymap.cpp
@@ -58,10 +59,6 @@ if(NOT ANDROID)
"${MT_NATIVE_FRAG_SPV}" "$<TARGET_FILE_DIR:${MT_NATIVE_TARGET}>/native.frag.spv")
endif()
target_link_libraries(${MT_NATIVE_TARGET} PRIVATE Vulkan::Vulkan SDL3::SDL3)
if(APPLE)
target_link_libraries(${MT_NATIVE_TARGET} PRIVATE
"-framework AudioToolbox")
endif()
if(TARGET port_platform AND TARGET mtpython)
target_compile_definitions(${MT_NATIVE_TARGET} PRIVATE MT_NATIVE_HAS_LIVE_CLIENT=1)
target_link_libraries(${MT_NATIVE_TARGET} PRIVATE port_platform)
@@ -90,3 +87,15 @@ if(ANDROID)
"${MT_PYTHON_STDLIB_ZIP}" "${MT_NATIVE_ANDROID_ASSETS}/python27.zip")
endif()
endif()
# audio_decode against synthetic .wav and every .wav/.mp3 of the real 40250 Client (pack + BGM).
if(NOT ANDROID AND TARGET port_platform AND BUILD_TESTING)
add_executable(host_audio_decode_test
"${PROJECT_SOURCE_DIR}/tests/host/audio_decode_test.cpp"
audio_decode.cpp)
target_include_directories(host_audio_decode_test PRIVATE
"${CMAKE_CURRENT_SOURCE_DIR}" "${CMAKE_CURRENT_SOURCE_DIR}/third_party")
target_link_libraries(host_audio_decode_test PRIVATE port_platform SDL3::SDL3)
add_test(NAME host.audio_decode COMMAND $<TARGET_FILE:host_audio_decode_test> "${MT_40250_CLIENT}")
set_tests_properties(host.audio_decode PROPERTIES SKIP_RETURN_CODE 77)
endif()
+1 -1
View File
@@ -54,7 +54,7 @@ vendored `stb_image.h` is upstream v2.30 (SHA-256
### Interactive Playable Modes
Run the full 40250 client interactively (infinite frame loop until window close, resizable SDL3 window with automatic Vulkan swapchain recreation and `PythonBoot::SetUISize` sync, full keyboard/IME text input, SDL hardware cursor built from the original cursor images, and SDL3 + `AudioToolbox` `.wav`/`.mp3` audio):
Run the full 40250 client interactively (infinite frame loop until window close, resizable SDL3 window with automatic Vulkan swapchain recreation and `PythonBoot::SetUISize` sync, full keyboard/IME text input, SDL hardware cursor built from the original cursor images, and SDL3 `.wav`/`.mp3` audio via `audio_decode.cpp` + dr_mp3, the same on every platform):
```sh
# Interactive outdoor map session (auto-login via loopback FakeLoginServer)
+106
View File
@@ -0,0 +1,106 @@
#include "audio_decode.h"
#include <SDL3/SDL.h>
#define DR_MP3_IMPLEMENTATION
#define DR_MP3_NO_STDIO
#include "dr_mp3.h"
#include <algorithm>
#include <cstring>
namespace mt_host {
namespace {
std::uint32_t read_u32(const std::uint8_t* p) { return p[0] | p[1] << 8 | p[2] << 16 | std::uint32_t(p[3]) << 24; }
std::uint16_t read_u16(const std::uint8_t* p) { return std::uint16_t(p[0] | p[1] << 8); }
// Converts `bytes` of `src` audio to the mixer format, appending at most max_frames frames to out.
bool convert(const SDL_AudioSpec& src, const std::uint8_t* bytes, std::size_t size, std::vector<float>& out,
std::size_t max_frames) {
const SDL_AudioSpec dst{SDL_AUDIO_F32, kMixChannels, kMixRate};
Uint8* converted = nullptr;
int converted_len = 0;
if (!SDL_ConvertAudioSamples(&src, bytes, int(size), &dst, &converted, &converted_len))
return false;
std::size_t frames = std::size_t(converted_len) / (sizeof(float) * kMixChannels);
if (max_frames && frames > max_frames) frames = max_frames;
out.resize(frames * kMixChannels);
std::memcpy(out.data(), converted, out.size() * sizeof(float));
SDL_free(converted);
return !out.empty();
}
// RIFF/WAVE: the "fmt " and "data" chunks (WAVE_FORMAT_PCM, _IEEE_FLOAT, or _EXTENSIBLE carrying one of them).
bool decode_wav(const std::uint8_t* data, std::size_t size, std::vector<float>& out, std::size_t max_frames) {
std::uint16_t tag = 0, channels = 0, bits = 0;
std::uint32_t rate = 0;
const std::uint8_t* pcm = nullptr;
std::size_t pcm_size = 0;
for (std::size_t pos = 12; pos + 8 <= size;) {
const std::uint32_t chunk = read_u32(data + pos + 4);
const std::uint8_t* body = data + pos + 8;
const std::size_t avail = size - (pos + 8);
if (!std::memcmp(data + pos, "fmt ", 4) && chunk >= 16 && avail >= 16) {
tag = read_u16(body);
channels = read_u16(body + 2);
rate = read_u32(body + 4);
bits = read_u16(body + 14);
if (tag == 0xFFFE && chunk >= 26 && avail >= 26) tag = read_u16(body + 24); // SubFormat GUID's first word
} else if (!std::memcmp(data + pos, "data", 4)) {
pcm = body;
pcm_size = std::min<std::size_t>(chunk, avail);
}
pos += 8 + std::size_t(chunk) + (chunk & 1);
}
if (!pcm || !channels || !rate) return false;
SDL_AudioFormat format;
if (tag == 1 && bits == 8) format = SDL_AUDIO_U8;
else if (tag == 1 && bits == 16) format = SDL_AUDIO_S16LE;
else if (tag == 1 && bits == 32) format = SDL_AUDIO_S32LE;
else if (tag == 3 && bits == 32) format = SDL_AUDIO_F32LE;
else if (tag == 1 && bits == 24) {
// SDL has no 24-bit format: widen to 32-bit.
std::vector<std::uint8_t> wide(pcm_size / 3 * 4);
for (std::size_t i = 0, o = 0; i + 3 <= pcm_size; i += 3, o += 4) {
wide[o] = 0;
std::memcpy(&wide[o + 1], pcm + i, 3);
}
const SDL_AudioSpec src{SDL_AUDIO_S32LE, channels, int(rate)};
return convert(src, wide.data(), wide.size(), out, max_frames);
} else return false;
const std::size_t frame_bytes = std::size_t(SDL_AUDIO_BYTESIZE(format)) * channels;
pcm_size -= pcm_size % frame_bytes;
if (!pcm_size) return false;
const SDL_AudioSpec src{format, channels, int(rate)};
return convert(src, pcm, pcm_size, out, max_frames);
}
bool decode_mp3(const std::uint8_t* data, std::size_t size, std::vector<float>& out, std::size_t max_frames) {
drmp3_config config{};
drmp3_uint64 frames = 0;
float* pcm = drmp3_open_memory_and_read_pcm_frames_f32(data, size, &config, &frames, nullptr);
if (!pcm) return false;
bool ok = false;
if (frames && config.channels && config.sampleRate) {
const SDL_AudioSpec src{SDL_AUDIO_F32, int(config.channels), int(config.sampleRate)};
ok = convert(src, reinterpret_cast<const std::uint8_t*>(pcm),
std::size_t(frames) * config.channels * sizeof(float), out, max_frames);
}
drmp3_free(pcm, nullptr);
return ok;
}
} // namespace
bool decode_audio(const std::uint8_t* data, std::size_t size, std::vector<float>& out, int max_seconds) {
out.clear();
if (!data || size < 16) return false;
const std::size_t max_frames = max_seconds > 0 ? std::size_t(max_seconds) * kMixRate : 0;
if (!std::memcmp(data, "RIFF", 4) && !std::memcmp(data + 8, "WAVE", 4))
return decode_wav(data, size, out, max_frames);
return decode_mp3(data, size, out, max_frames);
}
} // namespace mt_host
+19
View File
@@ -0,0 +1,19 @@
#pragma once
#include <cstddef>
#include <cstdint>
#include <vector>
namespace mt_host {
// The mixer's format: interleaved stereo float at 44.1 kHz.
inline constexpr int kMixRate = 44100;
inline constexpr int kMixChannels = 2;
// Decodes a whole .wav (PCM 8/16/24/32-bit, float 32-bit) or .mp3 file held in memory into the mixer's
// format, resampling and up/down-mixing with SDL_ConvertAudioSamples. The 40250 packs hold 16-bit PCM
// .wav at 22/32/44.1 kHz and MPEG-1 layer III .mp3 at 44.1/48 kHz. Returns false (out empty) when the
// data is neither, or is damaged. At most max_seconds are kept (0 = no limit).
bool decode_audio(const std::uint8_t* data, std::size_t size, std::vector<float>& out, int max_seconds = 300);
} // namespace mt_host
+55 -88
View File
@@ -1,37 +1,17 @@
#include "native_audio.h"
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
#include "audio_decode.h"
#include "texture_decode.h"
#include "platform/MilesLib/AudioCommands.h"
#include <algorithm>
#include <chrono>
#include <cmath>
#include <cstring>
#include <iostream>
namespace mt_host {
#ifdef __APPLE__
OSStatus NativeAudioEngine::mem_audio_read_proc(
void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount) {
const auto* mem = static_cast<const MemoryAudioBuffer*>(inClientData);
if (inPosition < 0 || static_cast<std::size_t>(inPosition) >= mem->size) {
*actualCount = 0;
return noErr;
}
const std::size_t avail = mem->size - static_cast<std::size_t>(inPosition);
const std::size_t to_read = std::min<std::size_t>(requestCount, avail);
std::memcpy(buffer, mem->data + inPosition, to_read);
*actualCount = static_cast<UInt32>(to_read);
return noErr;
}
SInt64 NativeAudioEngine::mem_audio_get_size_proc(void* inClientData) {
return static_cast<SInt64>(static_cast<const MemoryAudioBuffer*>(inClientData)->size);
}
#endif
NativeAudioEngine::NativeAudioEngine() {
if (SDL_InitSubSystem(SDL_INIT_AUDIO)) {
SDL_AudioSpec spec{};
@@ -108,28 +88,24 @@ void NativeAudioEngine::pump() {
break;
case AudioCommand::PlayMusic:
case AudioCommand::FadeInMusic: {
const auto* clip = get_clip(cmd.filename);
if (clip && !clip->empty()) {
bgm_.samples = clip;
bgm_.frame_cursor = 0;
bgm_.loop = true;
bgm_.filename = cmd.filename;
bgm_.stop_after_fade = false;
const float target = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f);
bgm_.target_volume = target;
if (cmd.type == AudioCommand::FadeInMusic) {
bgm_.volume = 0.0f;
bgm_.fade_step_per_frame = target / (44100.0f * 1.5f);
} else {
bgm_.volume = target;
bgm_.fade_step_per_frame = 0.0f;
}
++played_music_;
}
std::vector<std::uint8_t> bytes;
if (cmd.filename.empty() || !read_live_pack_texture(cmd.filename, bytes)) break;
if (pending_music_.valid()) superseded_music_.push_back(std::move(pending_music_));
// The pack is read here (CEterPackManager is not thread-safe); only the decode runs off-thread.
pending_music_ = std::async(std::launch::async, [bytes = std::move(bytes), name = cmd.filename]() -> Clip {
auto clip = std::make_shared<std::vector<float>>();
if (!decode_audio(bytes.data(), bytes.size(), *clip))
SDL_Log("audio: cannot decode %s (%zu bytes)", name.c_str(), bytes.size());
return clip;
});
pending_music_file_ = cmd.filename;
pending_music_fade_in_ = cmd.type == AudioCommand::FadeInMusic;
pending_music_volume_ = cmd.volume;
break;
}
case AudioCommand::FadeOutMusic:
case AudioCommand::FadeOutAllMusic:
pending_music_file_.clear(); // a track still decoding is dropped when it finishes
if (bgm_.samples) {
bgm_.target_volume = 0.0f;
bgm_.fade_step_per_frame = -std::max(bgm_.volume, 0.01f) / (44100.0f * 1.0f);
@@ -158,6 +134,19 @@ void NativeAudioEngine::pump() {
}
}
if (pending_music_.valid() &&
pending_music_.wait_for(std::chrono::seconds(0)) == std::future_status::ready) {
Clip clip = pending_music_.get();
if (!pending_music_file_.empty() && clip && !clip->empty())
start_music(std::move(clip), pending_music_file_, pending_music_fade_in_, pending_music_volume_);
pending_music_file_.clear();
}
superseded_music_.erase(
std::remove_if(superseded_music_.begin(), superseded_music_.end(), [](const std::future<Clip>& f) {
return f.wait_for(std::chrono::seconds(0)) == std::future_status::ready;
}),
superseded_music_.end());
if (!stream_) return;
const int queued_bytes = SDL_GetAudioStreamAvailable(stream_);
if (queued_bytes < 0) return;
@@ -197,7 +186,10 @@ void NativeAudioEngine::pump() {
};
if (bgm_.samples) {
if (!mix_voice(bgm_)) bgm_.samples = nullptr;
if (!mix_voice(bgm_)) {
bgm_.samples = nullptr;
bgm_clip_.reset();
}
}
for (auto it = voices_.begin(); it != voices_.end();) {
if (!mix_voice(*it)) it = voices_.erase(it);
@@ -208,58 +200,33 @@ void NativeAudioEngine::pump() {
SDL_PutAudioStreamData(stream_, mix_buffer_.data(), static_cast<int>(mix_buffer_.size() * sizeof(float)));
}
void NativeAudioEngine::start_music(Clip clip, const std::string& filename, bool fade_in, float volume) {
bgm_clip_ = std::move(clip);
bgm_.samples = bgm_clip_.get();
bgm_.frame_cursor = 0;
bgm_.loop = true;
bgm_.filename = filename;
bgm_.stop_after_fade = false;
const float target = music_volume_ * std::clamp(volume, 0.0f, 1.0f);
bgm_.target_volume = target;
if (fade_in) {
bgm_.volume = 0.0f;
bgm_.fade_step_per_frame = target / (44100.0f * 1.5f);
} else {
bgm_.volume = target;
bgm_.fade_step_per_frame = 0.0f;
}
++played_music_;
SDL_Log("audio: music %s (%.0f s)", filename.c_str(), double(bgm_clip_->size() / 2) / kMixRate);
}
const std::vector<float>* NativeAudioEngine::get_clip(const std::string& vpath) {
if (vpath.empty()) return nullptr;
auto [it, inserted] = clips_.try_emplace(vpath);
if (!inserted) return &it->second;
#ifdef __APPLE__
std::vector<std::uint8_t> bytes;
if (!read_live_pack_texture(vpath, bytes) || bytes.size() < 16)
return &it->second;
MemoryAudioBuffer mem{bytes.data(), bytes.size()};
AudioFileID audio_file = nullptr;
if (AudioFileOpenWithCallbacks(
&mem, mem_audio_read_proc, nullptr, mem_audio_get_size_proc, nullptr, 0, &audio_file) != noErr ||
!audio_file) {
return &it->second;
}
ExtAudioFileRef ext_file = nullptr;
if (ExtAudioFileWrapAudioFileID(audio_file, false, &ext_file) != noErr || !ext_file) {
AudioFileClose(audio_file);
return &it->second;
}
AudioStreamBasicDescription client_format{};
client_format.mSampleRate = 44100.0;
client_format.mFormatID = kAudioFormatLinearPCM;
client_format.mFormatFlags = kAudioFormatFlagIsFloat | kAudioFormatFlagIsPacked;
client_format.mBytesPerPacket = 8;
client_format.mFramesPerPacket = 1;
client_format.mBytesPerFrame = 8;
client_format.mChannelsPerFrame = 2;
client_format.mBitsPerChannel = 32;
if (ExtAudioFileSetProperty(
ext_file,
kExtAudioFileProperty_ClientDataFormat,
sizeof(client_format),
&client_format) == noErr) {
std::vector<float> chunk(4096 * 2);
while (true) {
UInt32 frame_count = 4096;
AudioBufferList buf_list{};
buf_list.mNumberBuffers = 1;
buf_list.mBuffers[0].mNumberChannels = 2;
buf_list.mBuffers[0].mDataByteSize = static_cast<UInt32>(chunk.size() * sizeof(float));
buf_list.mBuffers[0].mData = chunk.data();
if (ExtAudioFileRead(ext_file, &frame_count, &buf_list) != noErr || frame_count == 0)
break;
it->second.insert(it->second.end(), chunk.begin(), chunk.begin() + std::size_t(frame_count) * 2);
if (it->second.size() > 44100 * 2 * 300) // cap single decoded track at 5 minutes
break;
}
}
ExtAudioFileDispose(ext_file);
AudioFileClose(audio_file);
#endif
if (read_live_pack_texture(vpath, bytes) && !decode_audio(bytes.data(), bytes.size(), it->second))
SDL_Log("audio: cannot decode %s (%zu bytes)", vpath.c_str(), bytes.size());
return &it->second;
}
+13 -14
View File
@@ -2,12 +2,11 @@
#ifdef MT_NATIVE_HAS_LIVE_CLIENT
#include <SDL3/SDL.h>
#ifdef __APPLE__
#include <AudioToolbox/AudioToolbox.h>
#endif
#include <cstddef>
#include <cstdint>
#include <future>
#include <memory>
#include <string>
#include <unordered_map>
#include <vector>
@@ -16,10 +15,6 @@ namespace mt_host {
// Drains the MilesLib AudioCommand queue and mixes 2D/3D sounds and BGM into an SDL audio stream.
class NativeAudioEngine {
struct MemoryAudioBuffer {
const std::uint8_t* data = nullptr;
std::size_t size = 0;
};
struct Voice {
const std::vector<float>* samples = nullptr;
std::size_t frame_cursor = 0;
@@ -33,13 +28,6 @@ class NativeAudioEngine {
std::string filename;
};
#ifdef __APPLE__
static OSStatus mem_audio_read_proc(
void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount);
static SInt64 mem_audio_get_size_proc(void* inClientData);
#endif
public:
NativeAudioEngine();
@@ -48,12 +36,23 @@ public:
void pump();
private:
using Clip = std::shared_ptr<const std::vector<float>>;
const std::vector<float>* get_clip(const std::string& vpath);
void start_music(Clip clip, const std::string& filename, bool fade_in, float volume);
SDL_AudioStream* stream_ = nullptr;
std::unordered_map<std::string, std::vector<float>> clips_;
std::vector<Voice> voices_;
Voice bgm_{};
// BGM is not cached in clips_ (a 5-minute track is ~100 MB of float): only the playing track is kept, and
// a new one decodes on a worker thread while the old one keeps playing.
Clip bgm_clip_;
std::future<Clip> pending_music_;
std::vector<std::future<Clip>> superseded_music_; // a std::async future blocks in its destructor
std::string pending_music_file_;
bool pending_music_fade_in_ = false;
float pending_music_volume_ = 1.0f;
std::vector<float> mix_buffer_;
float sound_volume_ = 0.8f;
float music_volume_ = 0.5f;
+5430
View File
File diff suppressed because it is too large Load Diff
+1
View File
@@ -175,6 +175,7 @@ if(BUILD_TESTING AND CMAKE_SYSTEM_NAME STREQUAL CMAKE_HOST_SYSTEM_NAME)
if(NOT MT_40250_CLIENT)
set(MT_40250_CLIENT "${MT_40250_SOURCE}/../../Client")
endif()
set(MT_40250_CLIENT "${MT_40250_CLIENT}" PARENT_SCOPE) # also for src/host tests
add_executable(port_eterpack_test ${PROJECT_SOURCE_DIR}/tests/port/port_eterpack_test.cpp)
target_link_libraries(port_eterpack_test PRIVATE port_platform)
add_test(NAME port.eterpack COMMAND $<TARGET_FILE:port_eterpack_test> ${MT_40250_CLIENT})