host: cross-platform audio decode (Android has sound)

Replace the macOS-only AudioToolbox decoder with mt_host::decode_audio: a
RIFF/WAVE PCM parser plus vendored dr_mp3, resampled/remixed to 44.1 kHz
stereo float with SDL_ConvertAudioSamples. macOS and Android now share one
path; the AudioToolbox link is gone.

BGM is no longer cached in clips_ (a 5-minute track is ~100 MB of float):
only the playing track is kept, and a new one decodes on a worker thread
(the pack read stays on the main thread) so map changes do not stall.

host.audio_decode test: synthetic wav layouts + all 1578 pack .wav, 9 pack
.mp3 and 25 Client/BGM .mp3 decode (7060 s of audio in ~3.6 s on M2).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
shenlei
2026-09-29 21:22:11 +09:00
co-authored by Claude Opus 5.5
parent 75d2dd8545
commit eab5938072
11 changed files with 5889 additions and 107 deletions
+1
View File
@@ -67,6 +67,7 @@ src/host/live_client.cpp # run_live_client:40250 客户端
src/host/vulkan_window.cpp # SDL 窗口/事件 + Vulkan 渲染器(回放 Render3DDraw / UIRenderCommand) src/host/vulkan_window.cpp # SDL 窗口/事件 + Vulkan 渲染器(回放 Render3DDraw / UIRenderCommand)
src/host/frame_policy.cpp # 帧率策略(FrameRatePolicy) src/host/frame_policy.cpp # 帧率策略(FrameRatePolicy)
src/host/native_audio.cpp # MilesLib AudioCommand → SDL 音频流 src/host/native_audio.cpp # MilesLib AudioCommand → SDL 音频流
src/host/audio_decode.cpp # .wav(手写 PCM 解析)/.mp3(dr_mp3)→ 44.1 kHz 立体声 float,全平台同一路径
src/host/android_perf.h # Android 帧率、ADPF、电量/温度(JNI 到 MainActivity) src/host/android_perf.h # Android 帧率、ADPF、电量/温度(JNI 到 MainActivity)
android/ # Gradle 工程:SDLActivity 外壳 MainActivity、APK 打包 android/ # Gradle 工程:SDLActivity 外壳 MainActivity、APK 打包
``` ```
+13
View File
@@ -125,6 +125,19 @@ with those layers.
the shipped product's documentation must credit "Portions of this software are the shipped product's documentation must credit "Portions of this software are
copyright © The FreeType Project (www.freetype.org)". copyright © The FreeType Project (www.freetype.org)".
## Host single-header libraries (`src/host/third_party/`)
Only the native host (`src/host`) includes these; each has one implementation TU.
* **stb_image v2.30** (`stb_image.h`, implemented in `stb_image_impl.cpp`): TGA/PNG/JPG
UI and texture decode. Public domain or MIT.
* **dr_mp3 v0.7.4** (`dr_mp3.h`, implemented in `audio_decode.cpp` with
`DR_MP3_NO_STDIO`): the `.mp3` half of `mt_host::decode_audio` (pack sounds and
`Client/BGM`), replacing the macOS-only AudioToolbox decoder so Android has audio.
Fetched 2026-09-29 from `https://cdn.jsdelivr.net/gh/mackron/dr_libs@master/dr_mp3.h`,
sha256 `997b7ee18de6e6b81e2a83f1ea9fc62aef25c62b28d48db95635f49e65de0a2f`.
Public domain (Unlicense) or MIT-0; no attribution required.
## Rebuilding / fetching ## Rebuilding / fetching
There are no submodules: miniLZO, Crypto++, CPython and FreeType are vendored There are no submodules: miniLZO, Crypto++, CPython and FreeType are vendored
+13 -4
View File
@@ -14,6 +14,7 @@ add_custom_target(mt_native_shaders DEPENDS "${MT_NATIVE_VERT_SPV}" "${MT_NATIVE
set(MT_NATIVE_RENDER_SOURCES set(MT_NATIVE_RENDER_SOURCES
main.cpp main.cpp
audio_decode.cpp
frame_policy.cpp frame_policy.cpp
host_util.cpp host_util.cpp
input_keymap.cpp input_keymap.cpp
@@ -58,10 +59,6 @@ if(NOT ANDROID)
"${MT_NATIVE_FRAG_SPV}" "$<TARGET_FILE_DIR:${MT_NATIVE_TARGET}>/native.frag.spv") "${MT_NATIVE_FRAG_SPV}" "$<TARGET_FILE_DIR:${MT_NATIVE_TARGET}>/native.frag.spv")
endif() endif()
target_link_libraries(${MT_NATIVE_TARGET} PRIVATE Vulkan::Vulkan SDL3::SDL3) target_link_libraries(${MT_NATIVE_TARGET} PRIVATE Vulkan::Vulkan SDL3::SDL3)
if(APPLE)
target_link_libraries(${MT_NATIVE_TARGET} PRIVATE
"-framework AudioToolbox")
endif()
if(TARGET port_platform AND TARGET mtpython) if(TARGET port_platform AND TARGET mtpython)
target_compile_definitions(${MT_NATIVE_TARGET} PRIVATE MT_NATIVE_HAS_LIVE_CLIENT=1) target_compile_definitions(${MT_NATIVE_TARGET} PRIVATE MT_NATIVE_HAS_LIVE_CLIENT=1)
target_link_libraries(${MT_NATIVE_TARGET} PRIVATE port_platform) target_link_libraries(${MT_NATIVE_TARGET} PRIVATE port_platform)
@@ -90,3 +87,15 @@ if(ANDROID)
"${MT_PYTHON_STDLIB_ZIP}" "${MT_NATIVE_ANDROID_ASSETS}/python27.zip") "${MT_PYTHON_STDLIB_ZIP}" "${MT_NATIVE_ANDROID_ASSETS}/python27.zip")
endif() endif()
endif() endif()
# audio_decode against synthetic .wav and every .wav/.mp3 of the real 40250 Client (pack + BGM).
if(NOT ANDROID AND TARGET port_platform AND BUILD_TESTING)
add_executable(host_audio_decode_test
"${PROJECT_SOURCE_DIR}/tests/host/audio_decode_test.cpp"
audio_decode.cpp)
target_include_directories(host_audio_decode_test PRIVATE
"${CMAKE_CURRENT_SOURCE_DIR}" "${CMAKE_CURRENT_SOURCE_DIR}/third_party")
target_link_libraries(host_audio_decode_test PRIVATE port_platform SDL3::SDL3)
add_test(NAME host.audio_decode COMMAND $<TARGET_FILE:host_audio_decode_test> "${MT_40250_CLIENT}")
set_tests_properties(host.audio_decode PROPERTIES SKIP_RETURN_CODE 77)
endif()
+1 -1
View File
@@ -54,7 +54,7 @@ vendored `stb_image.h` is upstream v2.30 (SHA-256
### Interactive Playable Modes ### Interactive Playable Modes
Run the full 40250 client interactively (infinite frame loop until window close, resizable SDL3 window with automatic Vulkan swapchain recreation and `PythonBoot::SetUISize` sync, full keyboard/IME text input, SDL hardware cursor built from the original cursor images, and SDL3 + `AudioToolbox` `.wav`/`.mp3` audio): Run the full 40250 client interactively (infinite frame loop until window close, resizable SDL3 window with automatic Vulkan swapchain recreation and `PythonBoot::SetUISize` sync, full keyboard/IME text input, SDL hardware cursor built from the original cursor images, and SDL3 `.wav`/`.mp3` audio via `audio_decode.cpp` + dr_mp3, the same on every platform):
```sh ```sh
# Interactive outdoor map session (auto-login via loopback FakeLoginServer) # Interactive outdoor map session (auto-login via loopback FakeLoginServer)
+106
View File
@@ -0,0 +1,106 @@
#include "audio_decode.h"
#include <SDL3/SDL.h>
#define DR_MP3_IMPLEMENTATION
#define DR_MP3_NO_STDIO
#include "dr_mp3.h"
#include <algorithm>
#include <cstring>
namespace mt_host {
namespace {
std::uint32_t read_u32(const std::uint8_t* p) { return p[0] | p[1] << 8 | p[2] << 16 | std::uint32_t(p[3]) << 24; }
std::uint16_t read_u16(const std::uint8_t* p) { return std::uint16_t(p[0] | p[1] << 8); }
// Converts `bytes` of `src` audio to the mixer format, appending at most max_frames frames to out.
bool convert(const SDL_AudioSpec& src, const std::uint8_t* bytes, std::size_t size, std::vector<float>& out,
std::size_t max_frames) {
const SDL_AudioSpec dst{SDL_AUDIO_F32, kMixChannels, kMixRate};
Uint8* converted = nullptr;
int converted_len = 0;
if (!SDL_ConvertAudioSamples(&src, bytes, int(size), &dst, &converted, &converted_len))
return false;
std::size_t frames = std::size_t(converted_len) / (sizeof(float) * kMixChannels);
if (max_frames && frames > max_frames) frames = max_frames;
out.resize(frames * kMixChannels);
std::memcpy(out.data(), converted, out.size() * sizeof(float));
SDL_free(converted);
return !out.empty();
}
// RIFF/WAVE: the "fmt " and "data" chunks (WAVE_FORMAT_PCM, _IEEE_FLOAT, or _EXTENSIBLE carrying one of them).
bool decode_wav(const std::uint8_t* data, std::size_t size, std::vector<float>& out, std::size_t max_frames) {
std::uint16_t tag = 0, channels = 0, bits = 0;
std::uint32_t rate = 0;
const std::uint8_t* pcm = nullptr;
std::size_t pcm_size = 0;
for (std::size_t pos = 12; pos + 8 <= size;) {
const std::uint32_t chunk = read_u32(data + pos + 4);
const std::uint8_t* body = data + pos + 8;
const std::size_t avail = size - (pos + 8);
if (!std::memcmp(data + pos, "fmt ", 4) && chunk >= 16 && avail >= 16) {
tag = read_u16(body);
channels = read_u16(body + 2);
rate = read_u32(body + 4);
bits = read_u16(body + 14);
if (tag == 0xFFFE && chunk >= 26 && avail >= 26) tag = read_u16(body + 24); // SubFormat GUID's first word
} else if (!std::memcmp(data + pos, "data", 4)) {
pcm = body;
pcm_size = std::min<std::size_t>(chunk, avail);
}
pos += 8 + std::size_t(chunk) + (chunk & 1);
}
if (!pcm || !channels || !rate) return false;
SDL_AudioFormat format;
if (tag == 1 && bits == 8) format = SDL_AUDIO_U8;
else if (tag == 1 && bits == 16) format = SDL_AUDIO_S16LE;
else if (tag == 1 && bits == 32) format = SDL_AUDIO_S32LE;
else if (tag == 3 && bits == 32) format = SDL_AUDIO_F32LE;
else if (tag == 1 && bits == 24) {
// SDL has no 24-bit format: widen to 32-bit.
std::vector<std::uint8_t> wide(pcm_size / 3 * 4);
for (std::size_t i = 0, o = 0; i + 3 <= pcm_size; i += 3, o += 4) {
wide[o] = 0;
std::memcpy(&wide[o + 1], pcm + i, 3);
}
const SDL_AudioSpec src{SDL_AUDIO_S32LE, channels, int(rate)};
return convert(src, wide.data(), wide.size(), out, max_frames);
} else return false;
const std::size_t frame_bytes = std::size_t(SDL_AUDIO_BYTESIZE(format)) * channels;
pcm_size -= pcm_size % frame_bytes;
if (!pcm_size) return false;
const SDL_AudioSpec src{format, channels, int(rate)};
return convert(src, pcm, pcm_size, out, max_frames);
}
bool decode_mp3(const std::uint8_t* data, std::size_t size, std::vector<float>& out, std::size_t max_frames) {
drmp3_config config{};
drmp3_uint64 frames = 0;
float* pcm = drmp3_open_memory_and_read_pcm_frames_f32(data, size, &config, &frames, nullptr);
if (!pcm) return false;
bool ok = false;
if (frames && config.channels && config.sampleRate) {
const SDL_AudioSpec src{SDL_AUDIO_F32, int(config.channels), int(config.sampleRate)};
ok = convert(src, reinterpret_cast<const std::uint8_t*>(pcm),
std::size_t(frames) * config.channels * sizeof(float), out, max_frames);
}
drmp3_free(pcm, nullptr);
return ok;
}
} // namespace
bool decode_audio(const std::uint8_t* data, std::size_t size, std::vector<float>& out, int max_seconds) {
out.clear();
if (!data || size < 16) return false;
const std::size_t max_frames = max_seconds > 0 ? std::size_t(max_seconds) * kMixRate : 0;
if (!std::memcmp(data, "RIFF", 4) && !std::memcmp(data + 8, "WAVE", 4))
return decode_wav(data, size, out, max_frames);
return decode_mp3(data, size, out, max_frames);
}
} // namespace mt_host
+19
View File
@@ -0,0 +1,19 @@
#pragma once
#include <cstddef>
#include <cstdint>
#include <vector>
namespace mt_host {
// The mixer's format: interleaved stereo float at 44.1 kHz.
inline constexpr int kMixRate = 44100;
inline constexpr int kMixChannels = 2;
// Decodes a whole .wav (PCM 8/16/24/32-bit, float 32-bit) or .mp3 file held in memory into the mixer's
// format, resampling and up/down-mixing with SDL_ConvertAudioSamples. The 40250 packs hold 16-bit PCM
// .wav at 22/32/44.1 kHz and MPEG-1 layer III .mp3 at 44.1/48 kHz. Returns false (out empty) when the
// data is neither, or is damaged. At most max_seconds are kept (0 = no limit).
bool decode_audio(const std::uint8_t* data, std::size_t size, std::vector<float>& out, int max_seconds = 300);
} // namespace mt_host
+55 -88
View File
@@ -1,37 +1,17 @@
#include "native_audio.h" #include "native_audio.h"
#ifdef MT_NATIVE_HAS_LIVE_CLIENT #ifdef MT_NATIVE_HAS_LIVE_CLIENT
#include "audio_decode.h"
#include "texture_decode.h" #include "texture_decode.h"
#include "platform/MilesLib/AudioCommands.h" #include "platform/MilesLib/AudioCommands.h"
#include <algorithm> #include <algorithm>
#include <chrono>
#include <cmath> #include <cmath>
#include <cstring> #include <cstring>
#include <iostream>
namespace mt_host { namespace mt_host {
#ifdef __APPLE__
OSStatus NativeAudioEngine::mem_audio_read_proc(
void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount) {
const auto* mem = static_cast<const MemoryAudioBuffer*>(inClientData);
if (inPosition < 0 || static_cast<std::size_t>(inPosition) >= mem->size) {
*actualCount = 0;
return noErr;
}
const std::size_t avail = mem->size - static_cast<std::size_t>(inPosition);
const std::size_t to_read = std::min<std::size_t>(requestCount, avail);
std::memcpy(buffer, mem->data + inPosition, to_read);
*actualCount = static_cast<UInt32>(to_read);
return noErr;
}
SInt64 NativeAudioEngine::mem_audio_get_size_proc(void* inClientData) {
return static_cast<SInt64>(static_cast<const MemoryAudioBuffer*>(inClientData)->size);
}
#endif
NativeAudioEngine::NativeAudioEngine() { NativeAudioEngine::NativeAudioEngine() {
if (SDL_InitSubSystem(SDL_INIT_AUDIO)) { if (SDL_InitSubSystem(SDL_INIT_AUDIO)) {
SDL_AudioSpec spec{}; SDL_AudioSpec spec{};
@@ -108,28 +88,24 @@ void NativeAudioEngine::pump() {
break; break;
case AudioCommand::PlayMusic: case AudioCommand::PlayMusic:
case AudioCommand::FadeInMusic: { case AudioCommand::FadeInMusic: {
const auto* clip = get_clip(cmd.filename); std::vector<std::uint8_t> bytes;
if (clip && !clip->empty()) { if (cmd.filename.empty() || !read_live_pack_texture(cmd.filename, bytes)) break;
bgm_.samples = clip; if (pending_music_.valid()) superseded_music_.push_back(std::move(pending_music_));
bgm_.frame_cursor = 0; // The pack is read here (CEterPackManager is not thread-safe); only the decode runs off-thread.
bgm_.loop = true; pending_music_ = std::async(std::launch::async, [bytes = std::move(bytes), name = cmd.filename]() -> Clip {
bgm_.filename = cmd.filename; auto clip = std::make_shared<std::vector<float>>();
bgm_.stop_after_fade = false; if (!decode_audio(bytes.data(), bytes.size(), *clip))
const float target = music_volume_ * std::clamp(cmd.volume, 0.0f, 1.0f); SDL_Log("audio: cannot decode %s (%zu bytes)", name.c_str(), bytes.size());
bgm_.target_volume = target; return clip;
if (cmd.type == AudioCommand::FadeInMusic) { });
bgm_.volume = 0.0f; pending_music_file_ = cmd.filename;
bgm_.fade_step_per_frame = target / (44100.0f * 1.5f); pending_music_fade_in_ = cmd.type == AudioCommand::FadeInMusic;
} else { pending_music_volume_ = cmd.volume;
bgm_.volume = target;
bgm_.fade_step_per_frame = 0.0f;
}
++played_music_;
}
break; break;
} }
case AudioCommand::FadeOutMusic: case AudioCommand::FadeOutMusic:
case AudioCommand::FadeOutAllMusic: case AudioCommand::FadeOutAllMusic:
pending_music_file_.clear(); // a track still decoding is dropped when it finishes
if (bgm_.samples) { if (bgm_.samples) {
bgm_.target_volume = 0.0f; bgm_.target_volume = 0.0f;
bgm_.fade_step_per_frame = -std::max(bgm_.volume, 0.01f) / (44100.0f * 1.0f); bgm_.fade_step_per_frame = -std::max(bgm_.volume, 0.01f) / (44100.0f * 1.0f);
@@ -158,6 +134,19 @@ void NativeAudioEngine::pump() {
} }
} }
if (pending_music_.valid() &&
pending_music_.wait_for(std::chrono::seconds(0)) == std::future_status::ready) {
Clip clip = pending_music_.get();
if (!pending_music_file_.empty() && clip && !clip->empty())
start_music(std::move(clip), pending_music_file_, pending_music_fade_in_, pending_music_volume_);
pending_music_file_.clear();
}
superseded_music_.erase(
std::remove_if(superseded_music_.begin(), superseded_music_.end(), [](const std::future<Clip>& f) {
return f.wait_for(std::chrono::seconds(0)) == std::future_status::ready;
}),
superseded_music_.end());
if (!stream_) return; if (!stream_) return;
const int queued_bytes = SDL_GetAudioStreamAvailable(stream_); const int queued_bytes = SDL_GetAudioStreamAvailable(stream_);
if (queued_bytes < 0) return; if (queued_bytes < 0) return;
@@ -197,7 +186,10 @@ void NativeAudioEngine::pump() {
}; };
if (bgm_.samples) { if (bgm_.samples) {
if (!mix_voice(bgm_)) bgm_.samples = nullptr; if (!mix_voice(bgm_)) {
bgm_.samples = nullptr;
bgm_clip_.reset();
}
} }
for (auto it = voices_.begin(); it != voices_.end();) { for (auto it = voices_.begin(); it != voices_.end();) {
if (!mix_voice(*it)) it = voices_.erase(it); if (!mix_voice(*it)) it = voices_.erase(it);
@@ -208,58 +200,33 @@ void NativeAudioEngine::pump() {
SDL_PutAudioStreamData(stream_, mix_buffer_.data(), static_cast<int>(mix_buffer_.size() * sizeof(float))); SDL_PutAudioStreamData(stream_, mix_buffer_.data(), static_cast<int>(mix_buffer_.size() * sizeof(float)));
} }
void NativeAudioEngine::start_music(Clip clip, const std::string& filename, bool fade_in, float volume) {
bgm_clip_ = std::move(clip);
bgm_.samples = bgm_clip_.get();
bgm_.frame_cursor = 0;
bgm_.loop = true;
bgm_.filename = filename;
bgm_.stop_after_fade = false;
const float target = music_volume_ * std::clamp(volume, 0.0f, 1.0f);
bgm_.target_volume = target;
if (fade_in) {
bgm_.volume = 0.0f;
bgm_.fade_step_per_frame = target / (44100.0f * 1.5f);
} else {
bgm_.volume = target;
bgm_.fade_step_per_frame = 0.0f;
}
++played_music_;
SDL_Log("audio: music %s (%.0f s)", filename.c_str(), double(bgm_clip_->size() / 2) / kMixRate);
}
const std::vector<float>* NativeAudioEngine::get_clip(const std::string& vpath) { const std::vector<float>* NativeAudioEngine::get_clip(const std::string& vpath) {
if (vpath.empty()) return nullptr; if (vpath.empty()) return nullptr;
auto [it, inserted] = clips_.try_emplace(vpath); auto [it, inserted] = clips_.try_emplace(vpath);
if (!inserted) return &it->second; if (!inserted) return &it->second;
#ifdef __APPLE__
std::vector<std::uint8_t> bytes; std::vector<std::uint8_t> bytes;
if (!read_live_pack_texture(vpath, bytes) || bytes.size() < 16) if (read_live_pack_texture(vpath, bytes) && !decode_audio(bytes.data(), bytes.size(), it->second))
return &it->second; SDL_Log("audio: cannot decode %s (%zu bytes)", vpath.c_str(), bytes.size());
MemoryAudioBuffer mem{bytes.data(), bytes.size()};
AudioFileID audio_file = nullptr;
if (AudioFileOpenWithCallbacks(
&mem, mem_audio_read_proc, nullptr, mem_audio_get_size_proc, nullptr, 0, &audio_file) != noErr ||
!audio_file) {
return &it->second;
}
ExtAudioFileRef ext_file = nullptr;
if (ExtAudioFileWrapAudioFileID(audio_file, false, &ext_file) != noErr || !ext_file) {
AudioFileClose(audio_file);
return &it->second;
}
AudioStreamBasicDescription client_format{};
client_format.mSampleRate = 44100.0;
client_format.mFormatID = kAudioFormatLinearPCM;
client_format.mFormatFlags = kAudioFormatFlagIsFloat | kAudioFormatFlagIsPacked;
client_format.mBytesPerPacket = 8;
client_format.mFramesPerPacket = 1;
client_format.mBytesPerFrame = 8;
client_format.mChannelsPerFrame = 2;
client_format.mBitsPerChannel = 32;
if (ExtAudioFileSetProperty(
ext_file,
kExtAudioFileProperty_ClientDataFormat,
sizeof(client_format),
&client_format) == noErr) {
std::vector<float> chunk(4096 * 2);
while (true) {
UInt32 frame_count = 4096;
AudioBufferList buf_list{};
buf_list.mNumberBuffers = 1;
buf_list.mBuffers[0].mNumberChannels = 2;
buf_list.mBuffers[0].mDataByteSize = static_cast<UInt32>(chunk.size() * sizeof(float));
buf_list.mBuffers[0].mData = chunk.data();
if (ExtAudioFileRead(ext_file, &frame_count, &buf_list) != noErr || frame_count == 0)
break;
it->second.insert(it->second.end(), chunk.begin(), chunk.begin() + std::size_t(frame_count) * 2);
if (it->second.size() > 44100 * 2 * 300) // cap single decoded track at 5 minutes
break;
}
}
ExtAudioFileDispose(ext_file);
AudioFileClose(audio_file);
#endif
return &it->second; return &it->second;
} }
+13 -14
View File
@@ -2,12 +2,11 @@
#ifdef MT_NATIVE_HAS_LIVE_CLIENT #ifdef MT_NATIVE_HAS_LIVE_CLIENT
#include <SDL3/SDL.h> #include <SDL3/SDL.h>
#ifdef __APPLE__
#include <AudioToolbox/AudioToolbox.h>
#endif
#include <cstddef> #include <cstddef>
#include <cstdint> #include <cstdint>
#include <future>
#include <memory>
#include <string> #include <string>
#include <unordered_map> #include <unordered_map>
#include <vector> #include <vector>
@@ -16,10 +15,6 @@ namespace mt_host {
// Drains the MilesLib AudioCommand queue and mixes 2D/3D sounds and BGM into an SDL audio stream. // Drains the MilesLib AudioCommand queue and mixes 2D/3D sounds and BGM into an SDL audio stream.
class NativeAudioEngine { class NativeAudioEngine {
struct MemoryAudioBuffer {
const std::uint8_t* data = nullptr;
std::size_t size = 0;
};
struct Voice { struct Voice {
const std::vector<float>* samples = nullptr; const std::vector<float>* samples = nullptr;
std::size_t frame_cursor = 0; std::size_t frame_cursor = 0;
@@ -33,13 +28,6 @@ class NativeAudioEngine {
std::string filename; std::string filename;
}; };
#ifdef __APPLE__
static OSStatus mem_audio_read_proc(
void* inClientData, SInt64 inPosition, UInt32 requestCount, void* buffer, UInt32* actualCount);
static SInt64 mem_audio_get_size_proc(void* inClientData);
#endif
public: public:
NativeAudioEngine(); NativeAudioEngine();
@@ -48,12 +36,23 @@ public:
void pump(); void pump();
private: private:
using Clip = std::shared_ptr<const std::vector<float>>;
const std::vector<float>* get_clip(const std::string& vpath); const std::vector<float>* get_clip(const std::string& vpath);
void start_music(Clip clip, const std::string& filename, bool fade_in, float volume);
SDL_AudioStream* stream_ = nullptr; SDL_AudioStream* stream_ = nullptr;
std::unordered_map<std::string, std::vector<float>> clips_; std::unordered_map<std::string, std::vector<float>> clips_;
std::vector<Voice> voices_; std::vector<Voice> voices_;
Voice bgm_{}; Voice bgm_{};
// BGM is not cached in clips_ (a 5-minute track is ~100 MB of float): only the playing track is kept, and
// a new one decodes on a worker thread while the old one keeps playing.
Clip bgm_clip_;
std::future<Clip> pending_music_;
std::vector<std::future<Clip>> superseded_music_; // a std::async future blocks in its destructor
std::string pending_music_file_;
bool pending_music_fade_in_ = false;
float pending_music_volume_ = 1.0f;
std::vector<float> mix_buffer_; std::vector<float> mix_buffer_;
float sound_volume_ = 0.8f; float sound_volume_ = 0.8f;
float music_volume_ = 0.5f; float music_volume_ = 0.5f;
+5430
View File
File diff suppressed because it is too large Load Diff
+1
View File
@@ -175,6 +175,7 @@ if(BUILD_TESTING AND CMAKE_SYSTEM_NAME STREQUAL CMAKE_HOST_SYSTEM_NAME)
if(NOT MT_40250_CLIENT) if(NOT MT_40250_CLIENT)
set(MT_40250_CLIENT "${MT_40250_SOURCE}/../../Client") set(MT_40250_CLIENT "${MT_40250_SOURCE}/../../Client")
endif() endif()
set(MT_40250_CLIENT "${MT_40250_CLIENT}" PARENT_SCOPE) # also for src/host tests
add_executable(port_eterpack_test ${PROJECT_SOURCE_DIR}/tests/port/port_eterpack_test.cpp) add_executable(port_eterpack_test ${PROJECT_SOURCE_DIR}/tests/port/port_eterpack_test.cpp)
target_link_libraries(port_eterpack_test PRIVATE port_platform) target_link_libraries(port_eterpack_test PRIVATE port_platform)
add_test(NAME port.eterpack COMMAND $<TARGET_FILE:port_eterpack_test> ${MT_40250_CLIENT}) add_test(NAME port.eterpack COMMAND $<TARGET_FILE:port_eterpack_test> ${MT_40250_CLIENT})
+237
View File
@@ -0,0 +1,237 @@
// host/audio_decode against synthetic .wav layouts and the real 40250 audio: every .wav/.mp3 in Client/pack
// (through CEterPackManager, as the live client reads them) and every Client/BGM/*.mp3 must decode to
// non-empty 44.1 kHz stereo float whose length matches the source duration.
//
// host_audio_decode_test <40250 Client dir>
//
// The pack/BGM half exits 77 (ctest SKIP) when the client is missing, unless MT_ASSETS_STRICT=1.
#include "EterPack/StdAfx.h"
#include "EterPack/EterPackManager.h"
#include "../../src/platform/UserInterface/UserInterface.h"
#include "audio_decode.h"
#include <algorithm>
#include <chrono>
#include <cmath>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <fstream>
#include <iterator>
#include <string>
#include <vector>
#include <dirent.h>
#include <strings.h>
#include <unistd.h>
using mt_host::decode_audio;
static int g_failures = 0;
#define CHECK(cond) \
do { \
if (!(cond)) { \
std::fprintf(stderr, "%s:%d: CHECK(%s)\n", __FILE__, __LINE__, #cond); \
++g_failures; \
} \
} while (0)
static void put16(std::vector<std::uint8_t>& v, unsigned x) { v.push_back(x & 0xff); v.push_back(x >> 8 & 0xff); }
static void put32(std::vector<std::uint8_t>& v, unsigned x) { put16(v, x & 0xffff); put16(v, x >> 16); }
// A RIFF/WAVE file with a LIST chunk before "fmt " (odd-sized, so padded) and `frames` frames of a ramp.
static std::vector<std::uint8_t> make_wav(unsigned tag, unsigned channels, unsigned rate, unsigned bits,
unsigned frames, bool extensible = false)
{
std::vector<std::uint8_t> fmt;
put16(fmt, extensible ? 0xFFFE : tag);
put16(fmt, channels);
put32(fmt, rate);
put32(fmt, rate * channels * bits / 8);
put16(fmt, channels * bits / 8);
put16(fmt, bits);
if (extensible)
{
put16(fmt, 22);
put16(fmt, bits);
put32(fmt, 0);
put16(fmt, tag);
const std::uint8_t guid_tail[14] = {0x00, 0x00, 0x00, 0x00, 0x10, 0x00, 0x80, 0x00, 0x00, 0xAA, 0x00, 0x38, 0x9B, 0x71};
fmt.insert(fmt.end(), guid_tail, guid_tail + 14);
}
std::vector<std::uint8_t> pcm;
for (unsigned i = 0; i < frames * channels; ++i)
{
const double s = std::sin(i * 0.01) * 0.5;
if (bits == 8) pcm.push_back(std::uint8_t(128 + s * 127));
else if (bits == 16) put16(pcm, unsigned(std::int16_t(s * 32767)));
else if (bits == 24) { const unsigned x = unsigned(std::int32_t(s * 8388607)); put16(pcm, x); pcm.push_back(x >> 16 & 0xff); }
else if (tag == 3) { float f = float(s); unsigned x; std::memcpy(&x, &f, 4); put32(pcm, x); }
else put32(pcm, unsigned(std::int32_t(s * 2147483647.0)));
}
std::vector<std::uint8_t> out = {'R', 'I', 'F', 'F', 0, 0, 0, 0, 'W', 'A', 'V', 'E', 'L', 'I', 'S', 'T', 3, 0, 0, 0, 'a', 'b', 'c', 0};
out.insert(out.end(), {'f', 'm', 't', ' '});
put32(out, unsigned(fmt.size()));
out.insert(out.end(), fmt.begin(), fmt.end());
out.insert(out.end(), {'d', 'a', 't', 'a'});
put32(out, unsigned(pcm.size()));
out.insert(out.end(), pcm.begin(), pcm.end());
const unsigned riff = unsigned(out.size() - 8);
std::memcpy(&out[4], &riff, 4);
return out;
}
// Output frame count within a few frames of the resampled source length.
static bool frames_match(const std::vector<float>& out, double src_frames, double src_rate)
{
const double expected = src_frames * mt_host::kMixRate / src_rate;
return out.size() % 2 == 0 && std::fabs(double(out.size() / 2) - expected) <= 8 + expected * 0.001;
}
static bool peak_ok(const std::vector<float>& out)
{
float peak = 0;
for (float f : out)
{
if (!std::isfinite(f)) return false;
peak = std::max(peak, std::fabs(f));
}
return peak <= 1.5f;
}
static void synthetic()
{
struct Case { unsigned tag, channels, rate, bits; bool ext; };
const Case cases[] = {
{1, 1, 22050, 16, false}, {1, 1, 32000, 16, false}, {1, 1, 44100, 16, false}, {1, 2, 44100, 16, false},
{1, 1, 22000, 16, false}, {1, 1, 11025, 8, false}, {1, 2, 48000, 24, false}, {1, 2, 44100, 32, false},
{3, 1, 44100, 32, false}, {1, 2, 22050, 16, true}, {3, 2, 48000, 32, true},
};
for (const Case& c : cases)
{
const unsigned frames = c.rate / 2;
std::vector<float> out;
const std::vector<std::uint8_t> wav = make_wav(c.tag, c.channels, c.rate, c.bits, frames, c.ext);
const bool ok = decode_audio(wav.data(), wav.size(), out);
if (!ok || !frames_match(out, frames, c.rate) || !peak_ok(out))
std::fprintf(stderr, "synthetic tag %u ch %u %u Hz %u bit ext %d: ok %d frames %zu\n", c.tag, c.channels,
c.rate, c.bits, c.ext, ok, out.size() / 2);
CHECK(ok && frames_match(out, frames, c.rate) && peak_ok(out));
}
// Mono is duplicated to both channels.
{
std::vector<float> out;
const std::vector<std::uint8_t> wav = make_wav(1, 1, 44100, 16, 1000);
CHECK(decode_audio(wav.data(), wav.size(), out) && out.size() == 2000);
bool same = out.size() == 2000;
for (size_t i = 0; same && i < out.size(); i += 2)
same = out[i] == out[i + 1];
CHECK(same);
}
// The length cap.
{
std::vector<float> out;
const std::vector<std::uint8_t> wav = make_wav(1, 1, 44100, 16, 44100 * 3);
CHECK(decode_audio(wav.data(), wav.size(), out, 1) && out.size() == 44100 * 2);
}
// Garbage, truncation, unsupported formats.
{
std::vector<float> out;
std::vector<std::uint8_t> junk(4096, 0x5a);
CHECK(!decode_audio(junk.data(), junk.size(), out) && out.empty());
CHECK(!decode_audio(nullptr, 0, out));
std::vector<std::uint8_t> wav = make_wav(1, 1, 22050, 16, 1000);
wav.resize(40);
CHECK(!decode_audio(wav.data(), wav.size(), out));
const std::vector<std::uint8_t> adpcm = make_wav(2, 1, 22050, 16, 1000);
CHECK(!decode_audio(adpcm.data(), adpcm.size(), out));
}
}
static bool has_ext(const std::string& name, const char* ext)
{
const size_t n = std::strlen(ext);
return name.size() > n && strcasecmp(name.c_str() + name.size() - n, ext) == 0;
}
int main(int argc, char** argv)
{
synthetic();
const char* strict = std::getenv("MT_ASSETS_STRICT");
const bool is_strict = strict && std::string(strict) == "1";
const std::string client = argc > 1 ? argv[1] : "";
if (client.empty() || chdir(client.c_str()) != 0 || access("pack/Index", 0) != 0)
{
std::fprintf(stderr, "host_audio_decode_test: no 40250 Client/pack at '%s'\n", client.c_str());
if (g_failures) return 1;
return is_strict ? 1 : 77;
}
PackSingletons();
CHECK(PackInitialize("pack"));
struct Access : CEterPackManager
{
static const CEterFileDict::TDict& dict(CEterPackManager& m) { return static_cast<Access&>(m).m_FileDict.GetDict(); }
};
CEterPackManager& mgr = CEterPackManager::Instance();
std::vector<std::string> names;
for (const auto& kv : Access::dict(mgr))
{
const std::string name = kv.second.pkInfo->filename;
if ((has_ext(name, ".wav") || has_ext(name, ".mp3")) && (names.empty() || names.back() != name))
names.push_back(name);
}
std::sort(names.begin(), names.end());
names.erase(std::unique(names.begin(), names.end()), names.end());
size_t wav = 0, mp3 = 0, bgm = 0, failed = 0;
double seconds = 0, decode_ms = 0;
auto decode_one = [&](const std::string& label, const std::uint8_t* data, size_t size) {
std::vector<float> out;
const auto t0 = std::chrono::steady_clock::now();
const bool ok = decode_audio(data, size, out) && peak_ok(out);
decode_ms += std::chrono::duration<double, std::milli>(std::chrono::steady_clock::now() - t0).count();
if (!ok)
{
if (failed < 10) std::fprintf(stderr, "cannot decode %s (%zu bytes)\n", label.c_str(), size);
++failed;
}
seconds += out.size() / 2.0 / mt_host::kMixRate;
};
for (const std::string& name : names)
{
CMappedFile file;
LPCVOID data = nullptr;
if (!mgr.Get(file, name.c_str(), &data))
{
std::fprintf(stderr, "cannot read %s\n", name.c_str());
++failed;
continue;
}
(has_ext(name, ".wav") ? wav : mp3) += 1;
decode_one(name, static_cast<const std::uint8_t*>(data), file.Size());
}
if (DIR* d = opendir("BGM"))
{
while (dirent* e = readdir(d))
{
if (!has_ext(e->d_name, ".mp3")) continue;
std::ifstream in(std::string("BGM/") + e->d_name, std::ios::binary);
const std::vector<std::uint8_t> bytes((std::istreambuf_iterator<char>(in)), std::istreambuf_iterator<char>());
++bgm;
decode_one(std::string("BGM/") + e->d_name, bytes.data(), bytes.size());
}
closedir(d);
}
// 2026-09-29 survey (tools/epk_scan): 1601 PCM16 .wav entries (1578 distinct paths) and 9 .mp3 in the packs, 25 .mp3 in Client/BGM.
CHECK(wav == 1578);
CHECK(mp3 == 9);
CHECK(bgm == 25);
CHECK(failed == 0);
std::printf("pack wav %zu, pack mp3 %zu, BGM mp3 %zu, failed %zu, %.0f s of audio decoded in %.0f ms\n", wav, mp3,
bgm, failed, seconds, decode_ms);
if (g_failures)
std::fprintf(stderr, "%d failure(s)\n", g_failures);
return g_failures ? 1 : 0;
}