From 08b1c6d5623622ae1d7a07c16227a42b369eed4e Mon Sep 17 00:00:00 2001 From: Mike Solar Date: Sun, 7 Jun 2026 00:51:08 +0800 Subject: [PATCH] Implement audio sync foundations --- README.md | 10 +- app/audio/CMakeLists.txt | 6 + app/audio/audiolevelmeter.cpp | 103 ++++++++++++ app/audio/audiolevelmeter.h | 57 +++++++ app/audio/audiosynchronizer.cpp | 66 ++++++++ app/audio/audiosynchronizer.h | 55 +++++++ app/audio/audiowaveformsync.cpp | 152 ++++++++++++++++++ app/audio/audiowaveformsync.h | 58 +++++++ app/codec/CMakeLists.txt | 2 + app/codec/ffmpeg/ffmpegdecoder.cpp | 43 +++++ app/codec/timecodemetadata.cpp | 88 ++++++++++ app/codec/timecodemetadata.h | 48 ++++++ .../project/footage/footagedescription.cpp | 30 ++++ app/node/project/footage/footagedescription.h | 33 +++- app/widget/audiomonitor/audiomonitor.cpp | 10 +- docs/zh/README.md | 9 +- tests/gtest/CMakeLists.txt | 4 + tests/gtest/audio_level_meter_test.cpp | 80 +++++++++ tests/gtest/audio_synchronizer_test.cpp | 76 +++++++++ tests/gtest/audio_waveform_sync_test.cpp | 96 +++++++++++ tests/gtest/timecode_metadata_test.cpp | 86 ++++++++++ 21 files changed, 1094 insertions(+), 18 deletions(-) create mode 100644 app/audio/audiolevelmeter.cpp create mode 100644 app/audio/audiolevelmeter.h create mode 100644 app/audio/audiosynchronizer.cpp create mode 100644 app/audio/audiosynchronizer.h create mode 100644 app/audio/audiowaveformsync.cpp create mode 100644 app/audio/audiowaveformsync.h create mode 100644 app/codec/timecodemetadata.cpp create mode 100644 app/codec/timecodemetadata.h create mode 100644 tests/gtest/audio_level_meter_test.cpp create mode 100644 tests/gtest/audio_synchronizer_test.cpp create mode 100644 tests/gtest/audio_waveform_sync_test.cpp create mode 100644 tests/gtest/timecode_metadata_test.cpp diff --git a/README.md b/README.md index b5a961b1a..72ed901b7 100644 --- a/README.md +++ b/README.md @@ -25,11 +25,7 @@ See [`docs/build.md`](docs/build.md) for build instructions on Windows (MSYS2), | Version | Theme | Core Deliverables | Boundary Notes | |:--|:--|:--|:--| | **0.3** (Current) | **Plugin Architecture Milestone** | Production-ready OpenFX host support | Not about quantity of plugins, but "any OFX plugin loads without crashing" | -| **0.4** | **Color Grading & LUTs** | `.cube`/`.3dl` support, scopes (waveform/vectorscope/histogram), three-way color wheels | Olive 0.2 already has OpenColorIO foundation; this version adds UI wrappers and LUT entry points | -| **0.5** | **Audio Sync** | Waveform auto-sync (dual-system recording alignment), BWF timecode sync, audio meters (LUFS/VU) | Killer feature distinguishing Oak from Amber/Olive | -| **0.6** | **Proxy & Performance** | Proxy media workflow, hardware-accelerated export (NVENC/VideoToolbox), batch render queue | Solves 4K/8K usability | -| **0.7** | **Animation & Tracking** | Bézier keyframe curve editor, basic point tracking, image stabilizer | Olive 0.2 node system is well-suited for track-data-driven workflows | -| **0.8** | **Multicam & Collaboration** | Full multicam angle switching, OpenTimelineIO, EDL/XML import/export | Hand-off to/from other tools (Blender/Nuke/Resolve) | -| **0.9** | **Stability Milestone** | Project file format freeze (backward compatibility promise), crash recovery, autosave, memory optimization | "Feature freeze" testing period before 1.0 | +| **0.4** | **Color, Audio & Performance** | `.cube`/`.3dl` support, scopes (waveform/vectorscope/histogram), three-way color wheels, waveform auto-sync, BWF timecode sync, audio meters (LUFS/VU), proxy media workflow, hardware-accelerated export (NVENC/VideoToolbox), batch render queue | Combines the previous 0.4-0.6 scope into one usability milestone: color workflow, audio sync, and 4K/8K performance | +| **0.5** | **Animation, Tracking & Collaboration** | Bézier keyframe curve editor, basic point tracking, image stabilizer, full multicam angle switching, OpenTimelineIO, EDL/XML import/export | Combines the previous 0.7-0.8 scope into one timeline/interchange milestone | +| **0.6** | **Stability Milestone** | Project file format freeze (backward compatibility promise), crash recovery, autosave, memory optimization | "Feature freeze" testing period before 1.0 | | **1.0** | **Production Ready** | Complete documentation, installers, known issues list, community support channels | Declared "ready for serious projects" | - diff --git a/app/audio/CMakeLists.txt b/app/audio/CMakeLists.txt index 6b161d81b..a9b08de04 100644 --- a/app/audio/CMakeLists.txt +++ b/app/audio/CMakeLists.txt @@ -16,6 +16,12 @@ set(OLIVE_SOURCES ${OLIVE_SOURCES} + audio/audiolevelmeter.cpp + audio/audiolevelmeter.h + audio/audiosynchronizer.cpp + audio/audiosynchronizer.h + audio/audiowaveformsync.cpp + audio/audiowaveformsync.h audio/audiomanager.cpp audio/audiomanager.h audio/audioprocessor.cpp diff --git a/app/audio/audiolevelmeter.cpp b/app/audio/audiolevelmeter.cpp new file mode 100644 index 000000000..68f03d5b5 --- /dev/null +++ b/app/audio/audiolevelmeter.cpp @@ -0,0 +1,103 @@ +/*** + + Oak - Non-Linear Video Editor + Copyright (C) 2026 Oak Team + + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + + You should have received a copy of the GNU General Public License + along with this program. If not, see . + +***/ + +#include "audiolevelmeter.h" + +#include +#include + +#include "common/decibel.h" + +namespace olive +{ + +AudioLevelMeter::Stats +AudioLevelMeter::AnalyzeSampleBuffer(const core::SampleBuffer &samples) +{ + Stats stats; + + const int channel_count = samples.channel_count(); + const size_t sample_count = samples.sample_count(); + stats.channels.resize(channel_count); + + if (!channel_count || !sample_count) { + return stats; + } + + double total_square = 0.0; + size_t total_samples = 0; + + for (int channel = 0; channel < channel_count; channel++) { + const float *channel_data = samples.data(channel); + double peak = 0.0; + double square_sum = 0.0; + + for (size_t sample = 0; sample < sample_count; sample++) { + const double value = channel_data[sample]; + const double abs_value = std::abs(value); + + peak = std::max(peak, abs_value); + square_sum += value * value; + } + + const double mean_square = square_sum / static_cast(sample_count); + const double rms = std::sqrt(mean_square); + + ChannelStats channel_stats; + channel_stats.peak_linear = peak; + channel_stats.peak_db = LinearToDb(peak); + channel_stats.rms_linear = rms; + channel_stats.rms_db = LinearToDb(rms); + channel_stats.vu_db = channel_stats.rms_db; + stats.channels[channel] = channel_stats; + + stats.max_peak_linear = std::max(stats.max_peak_linear, peak); + total_square += square_sum; + total_samples += sample_count; + } + + stats.silence = qFuzzyIsNull(stats.max_peak_linear); + stats.integrated_lufs = PowerToLufs( + total_square / static_cast(total_samples)); + + return stats; +} + +double AudioLevelMeter::LinearToDb(double linear) +{ + if (linear <= 0.0) { + return Decibel::MINIMUM; + } + + return Decibel::fromLinear(linear); +} + +double AudioLevelMeter::PowerToLufs(double mean_square) +{ + if (mean_square <= 0.0) { + return Decibel::MINIMUM; + } + + // BS.1770 loudness uses K-weighted mean square. This first pass stores the + // compatible unit and can be extended with K-weighting without changing UI. + return -0.691 + 10.0 * std::log10(mean_square); +} + +} diff --git a/app/audio/audiolevelmeter.h b/app/audio/audiolevelmeter.h new file mode 100644 index 000000000..06d0442f8 --- /dev/null +++ b/app/audio/audiolevelmeter.h @@ -0,0 +1,57 @@ +/*** + + Oak - Non-Linear Video Editor + Copyright (C) 2026 Oak Team + + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + + You should have received a copy of the GNU General Public License + along with this program. If not, see . + +***/ + +#ifndef AUDIOLEVELMETER_H +#define AUDIOLEVELMETER_H + +#include + +#include "olive/core/render/samplebuffer.h" + +namespace olive +{ + +class AudioLevelMeter { +public: + struct ChannelStats { + double peak_linear = 0.0; + double peak_db = -200.0; + double rms_linear = 0.0; + double rms_db = -200.0; + double vu_db = -200.0; + }; + + struct Stats { + QVector channels; + double max_peak_linear = 0.0; + double integrated_lufs = -200.0; + bool silence = true; + }; + + static Stats AnalyzeSampleBuffer(const core::SampleBuffer &samples); + +private: + static double LinearToDb(double linear); + static double PowerToLufs(double mean_square); +}; + +} + +#endif // AUDIOLEVELMETER_H diff --git a/app/audio/audiosynchronizer.cpp b/app/audio/audiosynchronizer.cpp new file mode 100644 index 000000000..21c1ecb5e --- /dev/null +++ b/app/audio/audiosynchronizer.cpp @@ -0,0 +1,66 @@ +/*** + + Oak - Non-Linear Video Editor + Copyright (C) 2026 Oak Team + + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + + You should have received a copy of the GNU General Public License + along with this program. If not, see . + +***/ + +#include "audiosynchronizer.h" + +namespace olive +{ + +AudioSynchronizer::Placement AudioSynchronizer::PlaceBySourceTime( + const SourceClip &reference, const SourceClip &candidate, + const core::rational &reference_timeline_in) +{ + Placement placement; + if (!reference.has_source_start_time || + !candidate.has_source_start_time || + reference.source_start_time.isNaN() || + candidate.source_start_time.isNaN()) { + return placement; + } + + const core::rational reference_head_source = + reference.source_start_time + reference.media_in; + const core::rational candidate_head_source = + candidate.source_start_time + candidate.media_in; + + placement.timeline_in = + reference_timeline_in + candidate_head_source - reference_head_source; + placement.valid = !placement.timeline_in.isNaN(); + return placement; +} + +AudioSynchronizer::Placement AudioSynchronizer::PlaceByWaveformOffset( + const core::rational &reference_timeline_in, + int64_t candidate_offset_samples, int sample_rate) +{ + Placement placement; + if (sample_rate <= 0) { + return placement; + } + + placement.timeline_in = + reference_timeline_in + + core::rational::fromDouble(static_cast(candidate_offset_samples) / + static_cast(sample_rate)); + placement.valid = !placement.timeline_in.isNaN(); + return placement; +} + +} diff --git a/app/audio/audiosynchronizer.h b/app/audio/audiosynchronizer.h new file mode 100644 index 000000000..00dd28629 --- /dev/null +++ b/app/audio/audiosynchronizer.h @@ -0,0 +1,55 @@ +/*** + + Oak - Non-Linear Video Editor + Copyright (C) 2026 Oak Team + + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + + You should have received a copy of the GNU General Public License + along with this program. If not, see . + +***/ + +#ifndef AUDIOSYNCHRONIZER_H +#define AUDIOSYNCHRONIZER_H + +#include + +#include "olive/core/util/rational.h" + +namespace olive +{ + +class AudioSynchronizer { +public: + struct SourceClip { + core::rational source_start_time; + core::rational media_in; + bool has_source_start_time = false; + }; + + struct Placement { + core::rational timeline_in; + bool valid = false; + }; + + static Placement PlaceBySourceTime(const SourceClip &reference, + const SourceClip &candidate, + const core::rational &reference_timeline_in); + + static Placement PlaceByWaveformOffset( + const core::rational &reference_timeline_in, + int64_t candidate_offset_samples, int sample_rate); +}; + +} + +#endif // AUDIOSYNCHRONIZER_H diff --git a/app/audio/audiowaveformsync.cpp b/app/audio/audiowaveformsync.cpp new file mode 100644 index 000000000..075dd656f --- /dev/null +++ b/app/audio/audiowaveformsync.cpp @@ -0,0 +1,152 @@ +/*** + + Oak - Non-Linear Video Editor + Copyright (C) 2026 Oak Team + + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + + You should have received a copy of the GNU General Public License + along with this program. If not, see . + +***/ + +#include "audiowaveformsync.h" + +#include +#include + +namespace olive +{ + +QVector +AudioWaveformSync::ExtractRmsEnvelope(const core::SampleBuffer &samples, + size_t window_samples) +{ + QVector envelope; + + const int channel_count = samples.channel_count(); + const size_t sample_count = samples.sample_count(); + if (!channel_count || !sample_count || !window_samples) { + return envelope; + } + + const size_t window_count = + (sample_count + window_samples - 1) / window_samples; + envelope.resize(static_cast(window_count)); + + for (size_t window = 0; window < window_count; window++) { + const size_t start = window * window_samples; + const size_t end = std::min(start + window_samples, sample_count); + double square_sum = 0.0; + size_t total = 0; + + for (int channel = 0; channel < channel_count; channel++) { + const float *data = samples.data(channel); + for (size_t sample = start; sample < end; sample++) { + const double value = data[sample]; + square_sum += value * value; + total++; + } + } + + envelope[static_cast(window)] = + total ? std::sqrt(square_sum / static_cast(total)) : 0.0; + } + + return envelope; +} + +AudioWaveformSync::OffsetResult AudioWaveformSync::EstimateOffset( + const core::SampleBuffer &reference, const core::SampleBuffer &candidate, + size_t window_samples, int64_t max_offset_samples) +{ + if (!window_samples) { + return OffsetResult(); + } + + const QVector reference_envelope = + ExtractRmsEnvelope(reference, window_samples); + const QVector candidate_envelope = + ExtractRmsEnvelope(candidate, window_samples); + const int64_t max_offset_windows = + max_offset_samples / static_cast(window_samples); + + return EstimateEnvelopeOffset(reference_envelope, candidate_envelope, + window_samples, max_offset_windows); +} + +AudioWaveformSync::OffsetResult AudioWaveformSync::EstimateEnvelopeOffset( + const QVector &reference, const QVector &candidate, + size_t window_samples, int64_t max_offset_windows) +{ + OffsetResult result; + if (reference.isEmpty() || candidate.isEmpty() || !window_samples) { + return result; + } + + double best_score = -2.0; + int64_t best_lag = 0; + + for (int64_t lag = -max_offset_windows; lag <= max_offset_windows; lag++) { + const int reference_start = static_cast(std::max(0, -lag)); + const int candidate_start = static_cast(std::max(0, lag)); + const int overlap = std::min(reference.size() - reference_start, + candidate.size() - candidate_start); + + if (overlap < 2) { + continue; + } + + double reference_mean = 0.0; + double candidate_mean = 0.0; + for (int i = 0; i < overlap; i++) { + reference_mean += reference.at(reference_start + i); + candidate_mean += candidate.at(candidate_start + i); + } + reference_mean /= static_cast(overlap); + candidate_mean /= static_cast(overlap); + + double numerator = 0.0; + double reference_energy = 0.0; + double candidate_energy = 0.0; + for (int i = 0; i < overlap; i++) { + const double reference_value = + reference.at(reference_start + i) - reference_mean; + const double candidate_value = + candidate.at(candidate_start + i) - candidate_mean; + numerator += reference_value * candidate_value; + reference_energy += reference_value * reference_value; + candidate_energy += candidate_value * candidate_value; + } + + if (qFuzzyIsNull(reference_energy) || qFuzzyIsNull(candidate_energy)) { + continue; + } + + const double score = + numerator / std::sqrt(reference_energy * candidate_energy); + if (score > best_score) { + best_score = score; + best_lag = lag; + } + } + + if (best_score > -2.0) { + result.valid = true; + result.confidence = std::max(0.0, best_score); + result.offset_samples = + best_lag * static_cast(window_samples); + } + + return result; +} + +} diff --git a/app/audio/audiowaveformsync.h b/app/audio/audiowaveformsync.h new file mode 100644 index 000000000..35f3c1ecb --- /dev/null +++ b/app/audio/audiowaveformsync.h @@ -0,0 +1,58 @@ +/*** + + Oak - Non-Linear Video Editor + Copyright (C) 2026 Oak Team + + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + + You should have received a copy of the GNU General Public License + along with this program. If not, see . + +***/ + +#ifndef AUDIOWAVEFORMSYNC_H +#define AUDIOWAVEFORMSYNC_H + +#include + +#include + +#include "olive/core/render/samplebuffer.h" + +namespace olive +{ + +class AudioWaveformSync { +public: + struct OffsetResult { + int64_t offset_samples = 0; + double confidence = 0.0; + bool valid = false; + }; + + static QVector + ExtractRmsEnvelope(const core::SampleBuffer &samples, size_t window_samples); + + static OffsetResult EstimateOffset(const core::SampleBuffer &reference, + const core::SampleBuffer &candidate, + size_t window_samples, + int64_t max_offset_samples); + +private: + static OffsetResult EstimateEnvelopeOffset(const QVector &reference, + const QVector &candidate, + size_t window_samples, + int64_t max_offset_windows); +}; + +} + +#endif // AUDIOWAVEFORMSYNC_H diff --git a/app/codec/CMakeLists.txt b/app/codec/CMakeLists.txt index 608d82a08..a76089505 100644 --- a/app/codec/CMakeLists.txt +++ b/app/codec/CMakeLists.txt @@ -33,5 +33,7 @@ set(OLIVE_SOURCES codec/frame.h codec/planarfiledevice.cpp codec/planarfiledevice.h + codec/timecodemetadata.cpp + codec/timecodemetadata.h PARENT_SCOPE ) diff --git a/app/codec/ffmpeg/ffmpegdecoder.cpp b/app/codec/ffmpeg/ffmpegdecoder.cpp index 08df26241..64774046f 100644 --- a/app/codec/ffmpeg/ffmpegdecoder.cpp +++ b/app/codec/ffmpeg/ffmpegdecoder.cpp @@ -81,6 +81,7 @@ extern "C" { #include #include "codec/planarfiledevice.h" +#include "codec/timecodemetadata.h" #include "common/ffmpegutils.h" #include "common/filefunctions.h" #include "render/renderer.h" @@ -128,6 +129,36 @@ void DiscardSubtitleStreams(AVFormatContext *ctx) } } +TimecodeMetadata::SourceTime ExtractSourceStartTime( + AVDictionary *metadata, const rational &timebase, int sample_rate) +{ + if (!metadata) { + return TimecodeMetadata::SourceTime(); + } + + if (AVDictionaryEntry *entry = + av_dict_get(metadata, "timecode", nullptr, AV_DICT_IGNORE_SUFFIX)) { + TimecodeMetadata::SourceTime parsed = + TimecodeMetadata::FromTimecodeString( + QString::fromUtf8(entry->value), timebase); + if (parsed.valid) { + return parsed; + } + } + + if (AVDictionaryEntry *entry = av_dict_get( + metadata, "time_reference", nullptr, AV_DICT_IGNORE_SUFFIX)) { + TimecodeMetadata::SourceTime parsed = + TimecodeMetadata::FromBwfTimeReference( + QString::fromUtf8(entry->value), sample_rate); + if (parsed.valid) { + return parsed; + } + } + + return TimecodeMetadata::SourceTime(); +} + } // namespace FFmpegDecoder::FFmpegDecoder() @@ -530,6 +561,9 @@ FootageDescription FFmpegDecoder::Probe(const QString &filename, avformat_find_stream_info(fmt_ctx, nullptr); int64_t footage_duration = fmt_ctx->duration; + TimecodeMetadata::SourceTime source_start_time = + ExtractSourceStartTime(fmt_ctx->metadata, rational(1, AV_TIME_BASE), + 0); bool duration_guessed_from_bitrate = (fmt_ctx->duration_estimation_method == @@ -545,6 +579,11 @@ FootageDescription FFmpegDecoder::Probe(const QString &filename, for (unsigned int i = 0; i < fmt_ctx->nb_streams; i++) { // FFmpeg AVStream AVStream *avstream = fmt_ctx->streams[i]; + if (!source_start_time.valid) { + source_start_time = ExtractSourceStartTime( + avstream->metadata, avstream->time_base, + avstream->codecpar->sample_rate); + } // Find decoder for this stream, if it exists we can proceed const AVCodec *decoder = @@ -720,6 +759,10 @@ FootageDescription FFmpegDecoder::Probe(const QString &filename, } desc.SetStreamCount(fmt_ctx->nb_streams); + if (source_start_time.valid) { + desc.SetSourceStartTime(source_start_time.time, + source_start_time.source); + } if (video_streams == 0 && audio_streams > 0 && still_streams > 0) { // This footage has no video streams, but has audio and image streams. We've probably diff --git a/app/codec/timecodemetadata.cpp b/app/codec/timecodemetadata.cpp new file mode 100644 index 000000000..ebaa92088 --- /dev/null +++ b/app/codec/timecodemetadata.cpp @@ -0,0 +1,88 @@ +/*** + + Oak - Non-Linear Video Editor + Copyright (C) 2026 Oak Team + + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + + You should have received a copy of the GNU General Public License + along with this program. If not, see . + +***/ + +#include "timecodemetadata.h" + +#include +#include + +#include "olive/core/util/timecodefunctions.h" + +namespace olive +{ + +TimecodeMetadata::SourceTime TimecodeMetadata::FromTimecodeString( + const QString &timecode, const core::rational &timebase) +{ + SourceTime result; + const QString trimmed = timecode.trimmed(); + if (trimmed.isEmpty()) { + return result; + } + + bool ok = false; + const core::Timecode::Display display = + trimmed.contains(';') ? core::Timecode::kTimecodeDropFrame + : core::Timecode::kTimecodeNonDropFrame; + result.time = core::Timecode::timecode_to_time(trimmed.toStdString(), + timebase, display, &ok); + result.valid = ok; + if (ok) { + result.source = QStringLiteral("timecode"); + } + return result; +} + +TimecodeMetadata::SourceTime TimecodeMetadata::FromBwfTimeReference( + const QString &time_reference, int sample_rate) +{ + SourceTime result; + if (sample_rate <= 0) { + return result; + } + + bool ok = false; + const qulonglong samples = time_reference.trimmed().toULongLong(&ok); + if (!ok) { + return result; + } + + qulonglong numerator = samples; + qulonglong denominator = static_cast(sample_rate); + const qulonglong divisor = std::gcd(numerator, denominator); + numerator /= divisor; + denominator /= divisor; + + const qulonglong rational_limit = + static_cast(std::numeric_limits::max()); + if (numerator <= rational_limit && denominator <= rational_limit) { + result.time = + core::rational(static_cast(numerator), + static_cast(denominator)); + } else { + result.time = core::rational::fromDouble( + static_cast(samples) / static_cast(sample_rate)); + } + result.source = QStringLiteral("bwf_time_reference"); + result.valid = true; + return result; +} + +} diff --git a/app/codec/timecodemetadata.h b/app/codec/timecodemetadata.h new file mode 100644 index 000000000..c78a124a4 --- /dev/null +++ b/app/codec/timecodemetadata.h @@ -0,0 +1,48 @@ +/*** + + Oak - Non-Linear Video Editor + Copyright (C) 2026 Oak Team + + This program is free software: you can redistribute it and/or modify + it under the terms of the GNU General Public License as published by + the Free Software Foundation, either version 3 of the License, or + (at your option) any later version. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License for more details. + + You should have received a copy of the GNU General Public License + along with this program. If not, see . + +***/ + +#ifndef TIMECODEMETADATA_H +#define TIMECODEMETADATA_H + +#include + +#include "olive/core/util/rational.h" + +namespace olive +{ + +class TimecodeMetadata { +public: + struct SourceTime { + core::rational time; + QString source; + bool valid = false; + }; + + static SourceTime FromTimecodeString(const QString &timecode, + const core::rational &timebase); + + static SourceTime FromBwfTimeReference(const QString &time_reference, + int sample_rate); +}; + +} + +#endif // TIMECODEMETADATA_H diff --git a/app/node/project/footage/footagedescription.cpp b/app/node/project/footage/footagedescription.cpp index 06e82c8b7..66efe2370 100644 --- a/app/node/project/footage/footagedescription.cpp +++ b/app/node/project/footage/footagedescription.cpp @@ -63,6 +63,26 @@ bool FootageDescription::Load(const QString &filename) while (XMLReadNextStartElement(&reader)) { if (reader.name() == QStringLiteral("decoder")) { decoder_ = reader.readElementText(); + } else if (reader.name() == + QStringLiteral("sourcestarttime")) { + QString source; + { + XMLAttributeLoop((&reader), attr) + { + if (attr.name() == QStringLiteral("source")) { + source = attr.value().toString(); + } + } + } + + const QStringList split = + reader.readElementText().split('/'); + if (split.size() == 2) { + SetSourceStartTime( + rational(split.at(0).toInt(), + split.at(1).toInt()), + source); + } } else if (reader.name() == QStringLiteral("streams")) { { XMLAttributeLoop((&reader), attr) @@ -133,6 +153,16 @@ bool FootageDescription::Save(const QString &filename) const writer.writeTextElement(QStringLiteral("decoder"), decoder_); + if (has_source_start_time_) { + writer.writeStartElement(QStringLiteral("sourcestarttime")); + writer.writeAttribute(QStringLiteral("source"), + source_start_time_source_); + writer.writeCharacters(QStringLiteral("%1/%2").arg( + QString::number(source_start_time_.numerator()), + QString::number(source_start_time_.denominator()))); + writer.writeEndElement(); + } + writer.writeStartElement(QStringLiteral("streams")); writer.writeAttribute(QStringLiteral("count"), diff --git a/app/node/project/footage/footagedescription.h b/app/node/project/footage/footagedescription.h index 83589b622..f96d6d374 100644 --- a/app/node/project/footage/footagedescription.h +++ b/app/node/project/footage/footagedescription.h @@ -22,6 +22,8 @@ #ifndef FOOTAGEDESCRIPTION_H #define FOOTAGEDESCRIPTION_H +#include + #include "node/output/track/track.h" #include "render/subtitleparams.h" #include "render/videoparams.h" @@ -34,6 +36,7 @@ public: FootageDescription(const QString &decoder = QString()) : decoder_(decoder) , total_stream_count_(0) + , has_source_start_time_(false) { } @@ -131,6 +134,28 @@ public: total_stream_count_ = s; } + void SetSourceStartTime(const rational &time, const QString &source) + { + source_start_time_ = time; + source_start_time_source_ = source; + has_source_start_time_ = true; + } + + bool HasSourceStartTime() const + { + return has_source_start_time_; + } + + const rational &source_start_time() const + { + return source_start_time_; + } + + const QString &source_start_time_source() const + { + return source_start_time_source_; + } + bool Load(const QString &filename); bool Save(const QString &filename) const; @@ -163,7 +188,7 @@ public: } private: - static constexpr unsigned kFootageMetaVersion = 6; + static constexpr unsigned kFootageMetaVersion = 7; QString decoder_; @@ -174,6 +199,12 @@ private: QVector subtitle_streams_; int total_stream_count_; + + rational source_start_time_; + + QString source_start_time_source_; + + bool has_source_start_time_; }; } diff --git a/app/widget/audiomonitor/audiomonitor.cpp b/app/widget/audiomonitor/audiomonitor.cpp index 1df3c306d..8f01c060c 100644 --- a/app/widget/audiomonitor/audiomonitor.cpp +++ b/app/widget/audiomonitor/audiomonitor.cpp @@ -25,6 +25,7 @@ #include #include +#include "audio/audiolevelmeter.h" #include "audio/audiomanager.h" #include "common/decibel.h" #include "common/qtutils.h" @@ -89,10 +90,11 @@ void AudioMonitor::PushSampleBuffer(const SampleBuffer &d) QVector v(params_.channel_count(), 0); - AudioVisualWaveform::Sample summed = - AudioVisualWaveform::SumSamples(d, 0, d.sample_count()); - - AudioVisualWaveformSampleToInternalValues(summed, v); + const AudioLevelMeter::Stats stats = + AudioLevelMeter::AnalyzeSampleBuffer(d); + for (int i = 0; i < v.size() && i < stats.channels.size(); i++) { + v[i] = stats.channels.at(i).peak_linear; + } // Fill values because they get averaged out for smoothing values_.fill(v); diff --git a/docs/zh/README.md b/docs/zh/README.md index 0205689af..980742aee 100644 --- a/docs/zh/README.md +++ b/docs/zh/README.md @@ -37,10 +37,7 @@ Oak 视频编辑器是 Olive 的重命名分支,目标是打造更完善、更 | 版本 | 主题 | 核心交付物 | 边界说明 | |:--|:--|:--|:--| | **0.3**(当前) | **插件架构里程碑** | OpenFX 宿主支持完整可用 | 不追求插件数量,追求"任意 OFX 插件加载不崩溃" | -| **0.4** | **调色与 LUT** | `.cube`/`.3dl` 支持、示波器(波形/矢量/直方图)、三向色轮面板 | Olive 0.2 已有 OpenColorIO 基础,此版本做 UI 封装和 LUT 入口 | -| **0.5** | **音频同步** | 波形自动同步(双系统录音对齐)、BWF 时间码同步、音频表(LUFS/VU) | 这是 Oak 区别于 Amber/Olive 的杀手级功能 | -| **0.6** | **代理与性能** | 代理媒体工作流、硬件加速导出(NVENC/VideoToolbox)、批量渲染队列 | 解决 4K/8K 可用性问题 | -| **0.7** | **动画与跟踪** | 贝塞尔关键帧曲线编辑器、基础点跟踪、画面稳定器 | Olive 0.2 节点系统适合做跟踪数据驱动 | -| **0.8** | **多机位与协作** | 完整 Multicam 角度切换、OpenTimelineIO、EDL/XML 导入导出 | 与其他工具(Blender/Nuke/Resolve)交接 | -| **0.9** | **稳定性里程碑** | 项目文件格式冻结(向后兼容承诺)、崩溃恢复、Autosave、内存优化 | 1.0 前的"封版"测试期 | +| **0.4** | **调色、音频与性能** | `.cube`/`.3dl` 支持、示波器(波形/矢量/直方图)、三向色轮面板、波形自动同步(双系统录音对齐)、BWF 时间码同步、音频表(LUFS/VU)、代理媒体工作流、硬件加速导出(NVENC/VideoToolbox)、批量渲染队列 | 合并原 0.4-0.6 范围,集中解决调色工作流、音频同步和 4K/8K 可用性 | +| **0.5** | **动画、跟踪与协作** | 贝塞尔关键帧曲线编辑器、基础点跟踪、画面稳定器、完整 Multicam 角度切换、OpenTimelineIO、EDL/XML 导入导出 | 合并原 0.7-0.8 范围,集中处理时间线高级能力和外部工具交接 | +| **0.6** | **稳定性里程碑** | 项目文件格式冻结(向后兼容承诺)、崩溃恢复、Autosave、内存优化 | 1.0 前的"封版"测试期 | | **1.0** | **生产就绪** | 文档完整、安装包、已知问题清单、社区支持渠道 | 宣告"可用于严肃项目" | diff --git a/tests/gtest/CMakeLists.txt b/tests/gtest/CMakeLists.txt index 87b338477..2f79a09ac 100644 --- a/tests/gtest/CMakeLists.txt +++ b/tests/gtest/CMakeLists.txt @@ -3,6 +3,9 @@ add_executable(olive-gtest common_current_test.cpp common_xmlutils_test.cpp config_test.cpp + audio_level_meter_test.cpp + audio_synchronizer_test.cpp + audio_waveform_sync_test.cpp color_lut_test.cpp node_value_test.cpp node_keyframe_test.cpp @@ -38,6 +41,7 @@ add_executable(olive-gtest task_taskmanager_test.cpp module_smoke_test.cpp shader_resources_test.cpp + timecode_metadata_test.cpp timebased_widget_test.cpp timeline_coordinate_test.cpp timeline_workarea_test.cpp diff --git a/tests/gtest/audio_level_meter_test.cpp b/tests/gtest/audio_level_meter_test.cpp new file mode 100644 index 000000000..052c1b7a7 --- /dev/null +++ b/tests/gtest/audio_level_meter_test.cpp @@ -0,0 +1,80 @@ +#include + +#include + +#include "audio/audiolevelmeter.h" +#include "olive/core/render/audioparams.h" +#include "olive/core/render/samplebuffer.h" +#include "olive/core/render/sampleformat.h" + +extern "C" { +#include +} + +namespace { + +olive::core::AudioParams MakeStereoParams() +{ + return olive::core::AudioParams(48000, AV_CH_LAYOUT_STEREO, + olive::core::SampleFormat::F32P); +} + +} + +TEST(AudioLevelMeter, Silence) +{ + olive::core::SampleBuffer samples(MakeStereoParams(), size_t(480)); + samples.silence(); + + const olive::AudioLevelMeter::Stats stats = + olive::AudioLevelMeter::AnalyzeSampleBuffer(samples); + + ASSERT_EQ(stats.channels.size(), 2); + EXPECT_TRUE(stats.silence); + EXPECT_DOUBLE_EQ(stats.max_peak_linear, 0.0); + EXPECT_DOUBLE_EQ(stats.channels.at(0).peak_linear, 0.0); + EXPECT_DOUBLE_EQ(stats.channels.at(0).rms_linear, 0.0); + EXPECT_DOUBLE_EQ(stats.integrated_lufs, -200.0); +} + +TEST(AudioLevelMeter, ConstantSignal) +{ + olive::core::SampleBuffer samples(MakeStereoParams(), size_t(480)); + for (int channel = 0; channel < samples.channel_count(); channel++) { + float *data = samples.data(channel); + for (size_t i = 0; i < samples.sample_count(); i++) { + data[i] = 0.5f; + } + } + + const olive::AudioLevelMeter::Stats stats = + olive::AudioLevelMeter::AnalyzeSampleBuffer(samples); + + ASSERT_EQ(stats.channels.size(), 2); + EXPECT_FALSE(stats.silence); + EXPECT_DOUBLE_EQ(stats.max_peak_linear, 0.5); + EXPECT_NEAR(stats.channels.at(0).peak_db, -6.0206, 0.0001); + EXPECT_NEAR(stats.channels.at(0).rms_linear, 0.5, 0.0001); + EXPECT_NEAR(stats.channels.at(0).vu_db, -6.0206, 0.0001); + EXPECT_NEAR(stats.integrated_lufs, -6.7116, 0.0001); +} + +TEST(AudioLevelMeter, PerChannelPeaksAndRms) +{ + olive::core::SampleBuffer samples(MakeStereoParams(), size_t(4)); + float left[] = { 0.0f, 0.25f, -0.5f, 1.0f }; + float right[] = { 0.0f, -0.25f, 0.25f, -0.25f }; + samples.set(0, left, 4); + samples.set(1, right, 4); + + const olive::AudioLevelMeter::Stats stats = + olive::AudioLevelMeter::AnalyzeSampleBuffer(samples); + + ASSERT_EQ(stats.channels.size(), 2); + EXPECT_DOUBLE_EQ(stats.channels.at(0).peak_linear, 1.0); + EXPECT_DOUBLE_EQ(stats.channels.at(1).peak_linear, 0.25); + EXPECT_NEAR(stats.channels.at(0).rms_linear, + std::sqrt((0.25 * 0.25 + 0.5 * 0.5 + 1.0) / 4.0), 0.0001); + EXPECT_NEAR(stats.channels.at(1).rms_linear, + std::sqrt((0.25 * 0.25 * 3.0) / 4.0), 0.0001); +} diff --git a/tests/gtest/audio_synchronizer_test.cpp b/tests/gtest/audio_synchronizer_test.cpp new file mode 100644 index 000000000..5d3edce04 --- /dev/null +++ b/tests/gtest/audio_synchronizer_test.cpp @@ -0,0 +1,76 @@ +#include + +#include "audio/audiosynchronizer.h" + +TEST(AudioSynchronizer, PlacesCandidateBySourceStartTime) +{ + olive::AudioSynchronizer::SourceClip reference; + reference.source_start_time = olive::core::rational(100); + reference.has_source_start_time = true; + + olive::AudioSynchronizer::SourceClip candidate; + candidate.source_start_time = olive::core::rational(112); + candidate.has_source_start_time = true; + + const olive::AudioSynchronizer::Placement placement = + olive::AudioSynchronizer::PlaceBySourceTime( + reference, candidate, olive::core::rational(10)); + + ASSERT_TRUE(placement.valid); + EXPECT_EQ(placement.timeline_in, olive::core::rational(22)); +} + +TEST(AudioSynchronizer, AccountsForMediaInWhenPlacingBySourceTime) +{ + olive::AudioSynchronizer::SourceClip reference; + reference.source_start_time = olive::core::rational(100); + reference.media_in = olive::core::rational(2); + reference.has_source_start_time = true; + + olive::AudioSynchronizer::SourceClip candidate; + candidate.source_start_time = olive::core::rational(100); + candidate.media_in = olive::core::rational(5); + candidate.has_source_start_time = true; + + const olive::AudioSynchronizer::Placement placement = + olive::AudioSynchronizer::PlaceBySourceTime( + reference, candidate, olive::core::rational(20)); + + ASSERT_TRUE(placement.valid); + EXPECT_EQ(placement.timeline_in, olive::core::rational(23)); +} + +TEST(AudioSynchronizer, RejectsMissingSourceStartTime) +{ + olive::AudioSynchronizer::SourceClip reference; + reference.source_start_time = olive::core::rational(100); + reference.has_source_start_time = true; + + olive::AudioSynchronizer::SourceClip candidate; + + const olive::AudioSynchronizer::Placement placement = + olive::AudioSynchronizer::PlaceBySourceTime( + reference, candidate, olive::core::rational(10)); + + EXPECT_FALSE(placement.valid); +} + +TEST(AudioSynchronizer, PlacesCandidateByWaveformOffset) +{ + const olive::AudioSynchronizer::Placement placement = + olive::AudioSynchronizer::PlaceByWaveformOffset( + olive::core::rational(10), 24000, 48000); + + ASSERT_TRUE(placement.valid); + EXPECT_EQ(placement.timeline_in, olive::core::rational(21, 2)); +} + +TEST(AudioSynchronizer, SupportsCandidateLeadByWaveformOffset) +{ + const olive::AudioSynchronizer::Placement placement = + olive::AudioSynchronizer::PlaceByWaveformOffset( + olive::core::rational(10), -48000, 48000); + + ASSERT_TRUE(placement.valid); + EXPECT_EQ(placement.timeline_in, olive::core::rational(9)); +} diff --git a/tests/gtest/audio_waveform_sync_test.cpp b/tests/gtest/audio_waveform_sync_test.cpp new file mode 100644 index 000000000..3fe816590 --- /dev/null +++ b/tests/gtest/audio_waveform_sync_test.cpp @@ -0,0 +1,96 @@ +#include + +#include "audio/audiowaveformsync.h" +#include "olive/core/render/audioparams.h" +#include "olive/core/render/samplebuffer.h" +#include "olive/core/render/sampleformat.h" + +extern "C" { +#include +} + +namespace { + +olive::core::AudioParams MakeMonoParams() +{ + return olive::core::AudioParams(48000, AV_CH_LAYOUT_MONO, + olive::core::SampleFormat::F32P); +} + +olive::core::SampleBuffer MakeBuffer(const QVector &values) +{ + olive::core::SampleBuffer samples(MakeMonoParams(), + static_cast(values.size())); + float *data = samples.data(0); + for (int i = 0; i < values.size(); i++) { + data[i] = values.at(i); + } + return samples; +} + +} + +TEST(AudioWaveformSync, ExtractsRmsEnvelope) +{ + olive::core::SampleBuffer samples = + MakeBuffer({ 1.0f, -1.0f, 0.5f, -0.5f, 0.0f, 0.0f }); + + const QVector envelope = + olive::AudioWaveformSync::ExtractRmsEnvelope(samples, 2); + + ASSERT_EQ(envelope.size(), 3); + EXPECT_NEAR(envelope.at(0), 1.0, 0.0001); + EXPECT_NEAR(envelope.at(1), 0.5, 0.0001); + EXPECT_NEAR(envelope.at(2), 0.0, 0.0001); +} + +TEST(AudioWaveformSync, EstimatesCandidateLag) +{ + const QVector reference_values = { + 0.0f, 0.0f, 0.8f, 0.8f, 0.1f, 0.1f, 0.6f, 0.6f, 0.0f, 0.0f + }; + const QVector candidate_values = { + 0.0f, 0.0f, 0.0f, 0.0f, 0.8f, 0.8f, 0.1f, + 0.1f, 0.6f, 0.6f, 0.0f, 0.0f + }; + + const olive::AudioWaveformSync::OffsetResult result = + olive::AudioWaveformSync::EstimateOffset( + MakeBuffer(reference_values), MakeBuffer(candidate_values), 2, 8); + + ASSERT_TRUE(result.valid); + EXPECT_EQ(result.offset_samples, 2); + EXPECT_GT(result.confidence, 0.99); +} + +TEST(AudioWaveformSync, EstimatesCandidateLead) +{ + const QVector reference_values = { + 0.0f, 0.0f, 0.0f, 0.0f, 0.9f, 0.9f, 0.3f, + 0.3f, 0.7f, 0.7f, 0.0f, 0.0f + }; + const QVector candidate_values = { + 0.9f, 0.9f, 0.3f, 0.3f, 0.7f, 0.7f, 0.0f, 0.0f + }; + + const olive::AudioWaveformSync::OffsetResult result = + olive::AudioWaveformSync::EstimateOffset( + MakeBuffer(reference_values), MakeBuffer(candidate_values), 2, 4); + + ASSERT_TRUE(result.valid); + EXPECT_EQ(result.offset_samples, -4); + EXPECT_GT(result.confidence, 0.99); +} + +TEST(AudioWaveformSync, RejectsSilence) +{ + olive::core::SampleBuffer reference(MakeMonoParams(), size_t(16)); + olive::core::SampleBuffer candidate(MakeMonoParams(), size_t(16)); + reference.silence(); + candidate.silence(); + + const olive::AudioWaveformSync::OffsetResult result = + olive::AudioWaveformSync::EstimateOffset(reference, candidate, 4, 16); + + EXPECT_FALSE(result.valid); +} diff --git a/tests/gtest/timecode_metadata_test.cpp b/tests/gtest/timecode_metadata_test.cpp new file mode 100644 index 000000000..d646d39f2 --- /dev/null +++ b/tests/gtest/timecode_metadata_test.cpp @@ -0,0 +1,86 @@ +#include + +#include +#include + +#include "codec/timecodemetadata.h" +#include "node/project/footage/footagedescription.h" + +TEST(TimecodeMetadata, ParsesNonDropFrameTimecode) +{ + const olive::TimecodeMetadata::SourceTime parsed = + olive::TimecodeMetadata::FromTimecodeString( + QStringLiteral("01:02:03:12"), olive::core::rational(1, 24)); + + ASSERT_TRUE(parsed.valid); + EXPECT_EQ(parsed.source, QStringLiteral("timecode")); + EXPECT_EQ(parsed.time, olive::core::rational(1 * 3600 + 2 * 60 + 3, 1) + + olive::core::rational(12, 24)); +} + +TEST(TimecodeMetadata, ParsesDropFrameTimecode) +{ + const olive::TimecodeMetadata::SourceTime parsed = + olive::TimecodeMetadata::FromTimecodeString( + QStringLiteral("00:01:00;02"), olive::core::rational(1001, 30000)); + + ASSERT_TRUE(parsed.valid); + EXPECT_EQ(parsed.source, QStringLiteral("timecode")); + EXPECT_GT(parsed.time, olive::core::rational(59)); + EXPECT_LT(parsed.time, olive::core::rational(61)); +} + +TEST(TimecodeMetadata, ParsesBwfTimeReference) +{ + const olive::TimecodeMetadata::SourceTime parsed = + olive::TimecodeMetadata::FromBwfTimeReference( + QStringLiteral("96000"), 48000); + + ASSERT_TRUE(parsed.valid); + EXPECT_EQ(parsed.source, QStringLiteral("bwf_time_reference")); + EXPECT_EQ(parsed.time, olive::core::rational(2)); +} + +TEST(TimecodeMetadata, ParsesLargeBwfTimeReferenceWithoutTruncation) +{ + const olive::TimecodeMetadata::SourceTime parsed = + olive::TimecodeMetadata::FromBwfTimeReference( + QStringLiteral("4294967296"), 48000); + + ASSERT_TRUE(parsed.valid); + EXPECT_GT(parsed.time, olive::core::rational(89478)); + EXPECT_LT(parsed.time, olive::core::rational(89479)); +} + +TEST(TimecodeMetadata, RejectsInvalidMetadata) +{ + EXPECT_FALSE(olive::TimecodeMetadata::FromTimecodeString( + QString(), olive::core::rational(1, 24)) + .valid); + EXPECT_FALSE(olive::TimecodeMetadata::FromBwfTimeReference( + QStringLiteral("not-a-number"), 48000) + .valid); + EXPECT_FALSE(olive::TimecodeMetadata::FromBwfTimeReference( + QStringLiteral("123"), 0) + .valid); +} + +TEST(TimecodeMetadata, FootageDescriptionCachesSourceStartTime) +{ + QTemporaryDir dir; + ASSERT_TRUE(dir.isValid()); + const QString path = + QDir(dir.path()).filePath(QStringLiteral("footage-cache.xml")); + + olive::FootageDescription desc(QStringLiteral("ffmpeg")); + desc.SetSourceStartTime(olive::core::rational(96000, 48000), + QStringLiteral("bwf_time_reference")); + ASSERT_TRUE(desc.Save(path)); + + olive::FootageDescription loaded; + ASSERT_TRUE(loaded.Load(path)); + ASSERT_TRUE(loaded.HasSourceStartTime()); + EXPECT_EQ(loaded.source_start_time(), olive::core::rational(2)); + EXPECT_EQ(loaded.source_start_time_source(), + QStringLiteral("bwf_time_reference")); +}