diff --git a/README.md b/README.md index b5c2bb4..79cb8ee 100644 --- a/README.md +++ b/README.md @@ -7,6 +7,7 @@ GrowthLapse ist eine native macOS-App in SwiftUI, die aus mehreren Videos einen - macOS 13 oder neuer. - Swift 5.9 oder neuer mit macOS SDK, etwa über Xcode bzw. die Xcode Command Line Tools. - `ffmpeg` und `ffprobe` als ausführbare Dateien. Die App enthält diese Werkzeuge nicht. +- Für alle Funktionen empfiehlt sich unter Homebrew `brew install ffmpeg-full`; als Werkzeugpfade können `/opt/homebrew/opt/ffmpeg-full/bin/ffmpeg` und `/opt/homebrew/opt/ffmpeg-full/bin/ffprobe` ausgewählt werden. - Für VidStab: FFmpeg mit `vidstabdetect` und `vidstabtransform`; alternativ `deshake`. Für eingeblendete Clipnummern ist `drawtext` erforderlich. Eine Installation über Homebrew ist mit `brew install ffmpeg` möglich. Welche optionalen Filter dein Build enthält, zeigt `ffmpeg -hide_banner -filters`. diff --git a/Resources/Info.plist b/Resources/Info.plist index 36fa050..babaf98 100644 --- a/Resources/Info.plist +++ b/Resources/Info.plist @@ -19,9 +19,9 @@ CFBundlePackageType APPL CFBundleShortVersionString - 0.4.2 + 0.4.3 CFBundleVersion - 7 + 8 LSMinimumSystemVersion 13.0 NSHighResolutionCapable diff --git a/Sources/GrowthLapse/ContentView.swift b/Sources/GrowthLapse/ContentView.swift index 879fdfc..fa8d1cd 100644 --- a/Sources/GrowthLapse/ContentView.swift +++ b/Sources/GrowthLapse/ContentView.swift @@ -186,15 +186,8 @@ struct ContentView: View { .help("Stärkere Werte glätten mehr Bewegung. GrowthLapse hält den Sicherheits-Zoom dabei konstant, damit das Bild nicht pumpt.") .disabled(!settings.stabilizationEnabled) - Toggle("Auf Hintergrundpunkt fokussieren", isOn: $settings.stabilizationAnchorEnabled) - .help("Beschränkt nur die VidStab-Analyse auf einen Bereich um einen festen Hintergrundpunkt. Der Transform bleibt auf dem vollen Bild.") - .disabled(!settings.stabilizationEnabled || settings.stabilizationMethod != .vidstab) - - if settings.stabilizationAnchorEnabled { - percentSlider("Punkt X", value: $settings.stabilizationAnchorX, range: 0...1, help: "Horizontale Position des festen Hintergrundpunkts im Bild. 0% links, 100% rechts.") - percentSlider("Punkt Y", value: $settings.stabilizationAnchorY, range: 0...1, help: "Vertikale Position des festen Hintergrundpunkts im Bild. 0% oben, 100% unten.") - percentSlider("Analysebereich", value: $settings.stabilizationAnchorSize, range: 0.15...0.80, help: "Größe des Bereichs um den Hintergrundpunkt, den VidStab zur Bewegungsanalyse verwendet.") - } + Text("Kamerabewegung wird im ganzen Bild ermittelt. Der Gesichtsausschnitt bleibt innerhalb jedes Clips fest.") + .font(.caption).foregroundStyle(.secondary) } .disabled(processor.state.isRunning) @@ -214,13 +207,13 @@ struct ContentView: View { .help("Analysiert das Originalvideo vor dem Schneiden und sucht das Fenster, in dem Gesicht und beide Augen am besten sichtbar sind. Wenn nichts Brauchbares gefunden wird, nutzt GrowthLapse die normale Mitte/Start-Offset-Logik.") Toggle("Gemeinsame Gesichtsausrichtung", isOn: $settings.continuityAlignmentEnabled) - .help("Ermittelt für den gesamten Film eine gemeinsame Gesichtsgröße und Augenlinie. Unterschiedliche Kameraabstände werden durch den Ausschnitt ausgeglichen; fehlende Bildränder oder sehr kleine Gesichter werden markiert.") + .help("Verwendet deine Zielgröße ohne automatische Vergrößerung für Nahaufnahmen. Ein fester Ausschnitt mit Kopfrand hat Vorrang; zusätzlicher Gesichtszoom ist auf 12 % begrenzt.") .disabled(!settings.faceNormalizationEnabled) percentSlider("Augenlinie von oben", value: $settings.targetEyeY, range: 0.25...0.6, help: "Gemeinsame vertikale Position der Augen im Ergebnis. 40 % liegt etwas über der Bildmitte. Gilt nur, soweit genug Bildrand vorhanden ist.") .disabled(!settings.faceNormalizationEnabled) Toggle("Gesichtsgröße normalisieren", isOn: $settings.faceNormalizationEnabled) - .help("Verfolgt Gesicht und Augen mit Apple Vision und bewegt den Bildausschnitt geglättet mit. Das Bild bleibt immer vollflächig; fehlt Rand für die Zielgröße oder Drehung, hat ein randloses Bild Vorrang.") + .help("Nutzt Apple Vision für einen festen, zurückhaltenden Ausschnitt pro Clip. Bewegt den Ausschnitt nicht mit dem Kopf und dreht keine Kopfneigung gerade. Entwacklung und Platz um den Kopf haben Vorrang.") HStack { Text("Ziel-Gesichtshöhe") diff --git a/Sources/GrowthLapse/VideoProcessor.swift b/Sources/GrowthLapse/VideoProcessor.swift index 1751d57..7eef979 100644 --- a/Sources/GrowthLapse/VideoProcessor.swift +++ b/Sources/GrowthLapse/VideoProcessor.swift @@ -1320,6 +1320,7 @@ final class VideoProcessor: ObservableObject { let stabilizedURL = tempDirectory.appendingPathComponent(String(format: "stabilized_%04d.mp4", clipNumber)) try await extractSegment( source: video.url, + sourceIsHDR: sourceIsHDR, output: segmentURL, start: safeStart, duration: actualSegmentLength, @@ -1365,7 +1366,7 @@ final class VideoProcessor: ObservableObject { clipIndex: projectClip.index, ffmpeg: ffmpeg, sourceHasAudio: sourceHasAudio, - sourceIsHDR: sourceIsHDR, + sourceIsHDR: false, clipNumber: clipNumber, totalClips: videos.count ) @@ -1422,6 +1423,7 @@ final class VideoProcessor: ObservableObject { private func extractSegment( source: URL, + sourceIsHDR: Bool, output: URL, start: Double, duration: Double, @@ -1441,7 +1443,9 @@ final class VideoProcessor: ObservableObject { "-i", source.path, "-map", "0:v:0" ] - arguments.append(contentsOf: videoEncoderArguments(settings)) + // Tone-map at source precision, before any 8-bit encoder or stabilization. + arguments.append(contentsOf: ["-vf", sdrFilters(sourceIsHDR: sourceIsHDR).joined(separator: ",")]) + arguments.append(contentsOf: ["-c:v", "libx264", "-crf", "12", "-preset", "fast", "-pix_fmt", "yuv420p"]) if preserveAudio { arguments.append(contentsOf: ["-map", "0:a:0", "-c:a", "aac", "-b:a", "192k"]) } else { @@ -1545,7 +1549,7 @@ final class VideoProcessor: ObservableObject { progressText = "VidStab Analyse \(clipNumber) von \(totalClips)" appendLog("VidStab Analyse nur auf Segment \(segment.lastPathComponent) -> \(transformFileName)") if settings.stabilizationAnchorEnabled { - appendLog("VidStab Hintergrund-Anker: X \(Int(settings.stabilizationAnchorX * 100))%, Y \(Int(settings.stabilizationAnchorY * 100))%, Bereich \(Int(settings.stabilizationAnchorSize * 100))%") + appendLog("Alter Hintergrund-Anker wird durch Vollbildanalyse ersetzt; Analyse und Transform verwenden dieselben Bildkoordinaten.") } let detectArguments = [ "-y", @@ -1968,7 +1972,7 @@ final class VideoProcessor: ObservableObject { private func renderFingerprint(settings: RenderSettings, targetDimensions: VideoDimensions) -> String { [ - "continuity-v1", + "continuity-v2-sdr-first-static-framing", "\(settings.continuityAlignmentEnabled)", "\(settings.colorMatchingEnabled)", decimal(settings.colorMatchingStrength), @@ -2071,18 +2075,9 @@ final class VideoProcessor: ObservableObject { let strength = settings.stabilizationStrength let detect = "vidstabdetect=shakiness=\(strength.shakiness):accuracy=\(strength.accuracy):result=\(transformFileName)" - guard settings.stabilizationAnchorEnabled else { - return detect - } - - let width = max(2, evenDimension(Int(Double(video.width) * settings.stabilizationAnchorSize))) - let height = max(2, evenDimension(Int(Double(video.height) * settings.stabilizationAnchorSize))) - let centerX = Double(video.width) * settings.stabilizationAnchorX - let centerY = Double(video.height) * settings.stabilizationAnchorY - let x = evenDimension(Int(clamp(centerX - Double(width) / 2, min: 0, max: Double(max(0, video.width - width))))) - let y = evenDimension(Int(clamp(centerY - Double(height) / 2, min: 0, max: Double(max(0, video.height - height))))) - - return "crop=\(width):\(height):\(x):\(y),\(detect)" + // Detect and transform in the same full-frame coordinate system. + // Cropping detection alone gives rotations the wrong center on the full frame. + return detect } private func clamp(_ value: Double, min minimum: Double, max maximum: Double) -> Double { @@ -2099,7 +2094,7 @@ final class VideoProcessor: ObservableObject { sourceIsHDR: Bool, colorAdjustment: ColorAdjustment? = nil ) -> String { - var filters: [String] = [] + var filters = sdrFilters(sourceIsHDR: sourceIsHDR) if let alignment = faceAlignment { if abs(alignment.rotationRadians) > 0.001 { filters.append("rotate=\(decimal(alignment.rotationRadians)):ow=iw:oh=ih:fillcolor=black") @@ -2119,7 +2114,6 @@ final class VideoProcessor: ObservableObject { filters.append("scale=\(targetDimensions.width):\(targetDimensions.height):force_original_aspect_ratio=decrease") filters.append("pad=\(targetDimensions.width):\(targetDimensions.height):(ow-iw)/2:(oh-ih)/2") filters.append("setsar=1") - filters.append(contentsOf: sdrFilters(sourceIsHDR: sourceIsHDR)) if let colorAdjustment { filters.append(colorAdjustment.filter) } if burnInClipNumber { filters.append(clipNumberOverlayFilter(clipIndex)) @@ -2365,6 +2359,10 @@ final class VideoProcessor: ObservableObject { "zscale=primaries=bt709", "tonemap=tonemap=hable:desat=0", "zscale=transfer=bt709:matrix=bt709:range=tv", + "format=yuv444p16le", + // Tiny nonzero luma values avoid hqdn3d's automatic luma defaults. + "hqdn3d=0.0001:3:0.0001:4.5", + "zscale=dither=error_diffusion", "format=yuv420p", "setparams=color_primaries=bt709:color_trc=bt709:colorspace=bt709" ] @@ -3174,20 +3172,18 @@ private enum FaceAlignmentAnalyzer { } let outputAspect = Double(targetDimensions.width) / Double(targetDimensions.height) - var cropHeight = faceHeightPixels / targetFaceHeightRatio - var cropWidth = cropHeight * outputAspect - if cropWidth > Double(imageWidth) { - cropWidth = Double(imageWidth) - cropHeight = cropWidth / outputAspect - } - if cropHeight > Double(imageHeight) { - cropHeight = Double(imageHeight) - cropWidth = cropHeight * outputAspect - } + let fullHeight = min(Double(imageHeight), Double(imageWidth) / outputAspect) + let headTop = max(0, samples.map { (1 - $0.rect.maxY - $0.rect.height * 0.4) * Double(imageHeight) }.min() ?? 0) + let headBottom = min(Double(imageHeight), samples.map { (1 - $0.rect.minY + $0.rect.height * 0.12) * Double(imageHeight) }.max() ?? Double(imageHeight)) + let headLeft = max(0, samples.map { ($0.rect.minX - $0.rect.width * 0.2) * Double(imageWidth) }.min() ?? 0) + let headRight = min(Double(imageWidth), samples.map { ($0.rect.maxX + $0.rect.width * 0.2) * Double(imageWidth) }.max() ?? Double(imageWidth)) + let neededHeight = max(headBottom - headTop, (headRight - headLeft) / outputAspect) + let cropHeight = min(fullHeight, max(fullHeight / 1.12, faceHeightPixels / targetFaceHeightRatio, neededHeight)) + let cropWidth = cropHeight * outputAspect let evenCropWidth = max(2, even(Int(cropWidth.rounded(.down)))) let evenCropHeight = max(2, even(Int(cropHeight.rounded(.down)))) let eyeAngles = samples.compactMap(\.eyeAngle) - let proposedRotation = clamp(-median(eyeAngles), min: -0.14, max: 0.14) + let proposedRotation: Double = 0 // Head tilt is not camera shake; leave leveling to stabilization. let safeFrame = safeFrameAfterRotation( imageWidth: imageWidth, imageHeight: imageHeight, @@ -3226,13 +3222,17 @@ private enum FaceAlignmentAnalyzer { y: clamp(anchor.y - Double(evenCropHeight) * desiredY, min: minY, max: maxY) ) } - let positions = smoothed( - rawPositions, - minX: minX, - maxX: maxX, - minY: minY, - maxY: maxY - ) + // A constant crop does not chase head movement or undo camera stabilization. + let headMinX = max(minX, headRight - Double(evenCropWidth)) + let headMaxX = min(maxX, headLeft) + let headMinY = max(minY, headBottom - Double(evenCropHeight)) + let headMaxY = min(maxY, headTop) + let preferredX = median(rawPositions.map(\.x)) + let preferredY = median(rawPositions.map(\.y)) + let fixedX = headMinX <= headMaxX ? clamp(preferredX, min: headMinX, max: headMaxX) : clamp(preferredX, min: minX, max: maxX) + let fixedY = headMinY <= headMaxY ? clamp(preferredY, min: headMinY, max: headMaxY) : clamp(preferredY, min: minY, max: maxY) + boundaryLimited = boundaryLimited || headMinX > headMaxX || headMinY > headMaxY + let positions = [FaceAlignmentSample(time: 0, x: fixedX, y: fixedY)] return FaceAlignment( sourceWidth: imageWidth, diff --git a/Sources/GrowthLapse/VisualContinuity.swift b/Sources/GrowthLapse/VisualContinuity.swift index 5a96d36..f356708 100644 --- a/Sources/GrowthLapse/VisualContinuity.swift +++ b/Sources/GrowthLapse/VisualContinuity.swift @@ -189,16 +189,8 @@ enum ContinuityPlanner { static func sharedProfile(clips: [GrowthLapseProjectClip], dimensions: VideoDimensions, requested: Double, eyeY: Double) -> ContinuityProfile { - let aspect = Double(dimensions.width) / Double(dimensions.height) - let required = clips.compactMap { clip -> Double? in - let poses = clip.visualAnalysis?.window(start: clip.segmentStart, length: clip.segmentLength).compactMap(\.pose) ?? [] - guard !poses.isEmpty else { return nil } - let sourceAspect = Double(clip.width) / Double(clip.height) - // Reserve a small border for leveling the eye line. Outliers are flagged, not imposed on all clips. - return median(poses.map(\.height)) / (min(1, sourceAspect / aspect) * 0.94) - }.sorted() - let feasible = required.isEmpty ? requested : required[min(required.count - 1, Int(ceil(Double(required.count - 1) * 0.8)))] - return ContinuityProfile(faceHeightRatio: max(requested, min(requested + 0.12, feasible)), eyeY: eyeY) + // Framing must not enlarge every clip to accommodate close-up outliers. + return ContinuityProfile(faceHeightRatio: requested, eyeY: eyeY) } } diff --git a/Tests/Integration/ContinuityChecks.swift b/Tests/Integration/ContinuityChecks.swift index fc711ba..d3c2287 100644 --- a/Tests/Integration/ContinuityChecks.swift +++ b/Tests/Integration/ContinuityChecks.swift @@ -52,10 +52,10 @@ enum ContinuityChecks { } let profile = ContinuityPlanner.sharedProfile(clips: [clip(height: 0.08), clip(height: 0.35)], dimensions: VideoDimensions(width: 1920, height: 1080), requested: 0.26, eyeY: 0.4) - try check(profile.faceHeightRatio > 0.35 && profile.faceHeightRatio <= 0.38, "Different camera distances were not negotiated") + try check(profile.faceHeightRatio == 0.26, "Camera distances override requested framing") let extreme = ContinuityPlanner.sharedProfile(clips: [clip(height: 0.08), clip(height: 0.9)], dimensions: VideoDimensions(width: 1920, height: 1080), requested: 0.26, eyeY: 0.4) - try check(extreme.faceHeightRatio <= 0.38, "One extreme close-up forces excessive zoom on all clips") + try check(extreme.faceHeightRatio == 0.26, "One extreme close-up forces excessive zoom on all clips") print("PASS joint transition planning, sparse-window handling, face-only bounded colors and camera-distance profile") } diff --git a/Tests/Integration/RenderIntegration.swift b/Tests/Integration/RenderIntegration.swift index 6111d2a..ee11496 100644 --- a/Tests/Integration/RenderIntegration.swift +++ b/Tests/Integration/RenderIntegration.swift @@ -35,6 +35,31 @@ struct RenderIntegration { throw TestFailure.failed("Real project validation failed: \(validator.state)") } print("PASS real project load: \(project.clips.count) clips") + if ProcessInfo.processInfo.environment["GROWTHLAPSE_TEST_QUALITY"] == "1" { + let original = try Data(contentsOf: URL(fileURLWithPath: path)) + var subset = project + let prefixes = ["035", "062", "184"] + subset.clips = project.clips.filter { c in prefixes.contains { c.displayName.hasPrefix($0) } } + try require(subset.clips.count == 3, "Expected three visual regression clips") + validator.project = subset + var settings = SettingsStore.load() + let folder = FileManager.default.temporaryDirectory.appendingPathComponent("GrowthLapse-quality-\(UUID().uuidString)") + try FileManager.default.createDirectory(at: folder, withIntermediateDirectories: true) + settings.outputFile = folder.appendingPathComponent("Vergleich-neu.mp4") + settings.keepIntermediateClips = true + settings.keepRenderCacheForReview = true + validator.rerenderProject(settings: settings) + let deadline = Date().addingTimeInterval(300) + while validator.state.isRunning && Date() < deadline { try await Task.sleep(nanoseconds: 100_000_000) } + guard case .finished(let output) = validator.state else { + validator.cancel() + throw TestFailure.failed("Quality render: \(validator.state)\n\(validator.logs.suffix(25).joined(separator: "\n"))") + } + let after = try Data(contentsOf: URL(fileURLWithPath: path)) + try require(original == after, "Quality check changed user's project") + print("PASS real full-resolution quality render: \(output.path)") + return + } if ProcessInfo.processInfo.environment["GROWTHLAPSE_TEST_REAL_DRAFT"] == "1" { let original = try Data(contentsOf: URL(fileURLWithPath: path)) var subset = project @@ -380,6 +405,34 @@ struct RenderIntegration { try require(!manager.fileExists(atPath: goodProject.cacheDirectoryPath), "Owned cache was not deleted") try require(manager.fileExists(atPath: goodProject.outputFilePath), "Confirm deleted output") print("PASS cancelled rerender preserves old cache and confirm only deletes owned cache") + // HDR must become SDR before the first stabilization intermediate. + let hdrInput = root.appendingPathComponent("hdr", isDirectory: true) + try manager.createDirectory(at: hdrInput, withIntermediateDirectories: true) + let hdrSource = hdrInput.appendingPathComponent("hdr.mkv") + let hdrFixture = try await service.run(executable: tools.ffmpeg, arguments: [ + "-v", "error", "-y", "-f", "lavfi", "-i", "color=c=gray:s=320x240:r=15:d=1", + "-pix_fmt", "yuv420p10le", "-c:v", "ffv1", "-color_primaries", "bt2020", + "-color_trc", "arib-std-b67", "-colorspace", "bt2020nc", hdrSource.path]) + try require(hdrFixture.exitCode == 0, "HDR fixture generation failed") + var hdrSettings = RenderSettings() + hdrSettings.inputFolder = hdrInput + hdrSettings.outputFile = root.appendingPathComponent("HDR.mp4") + hdrSettings.outputFormat = .keepOriginal + hdrSettings.videoEncoder = .x264 + hdrSettings.segmentLength = 1 + hdrSettings.targetClipLength = 1 + hdrSettings.transitionLength = 0 + hdrSettings.stabilizationEnabled = true + hdrSettings.keepIntermediateClips = true + let hdrProcessor = VideoProcessor() + hdrProcessor.start(settings: hdrSettings) + try await wait(hdrProcessor) + let intermediate = hdrProcessor.project!.cacheDirectoryURL.appendingPathComponent("segment_0001.mp4") + let hdrProbe = try await probe(intermediate, tools: service, ffprobe: tools.ffprobe) + let hdrStreams = hdrProbe["streams"] as! [[String: Any]] + try require(hdrStreams[0]["color_transfer"] as? String == "bt709" + && hdrStreams[0]["color_primaries"] as? String == "bt709", "HDR persisted into 8-bit stabilization intermediate") + print("PASS 10-bit HLG source converted to SDR before stabilization encoding") print("All integration checks passed.") } } diff --git a/docs/VISUELLE_KONTINUITAET.md b/docs/VISUELLE_KONTINUITAET.md index 120befa..897ea65 100644 --- a/docs/VISUELLE_KONTINUITAET.md +++ b/docs/VISUELLE_KONTINUITAET.md @@ -1,3 +1,16 @@ +# Korrektur der Bildverarbeitung – Version 0.4.3 + +Die Prüfung realer Aufnahmen hat Grenzen der ursprünglichen Kontinuitätsautomatik gezeigt. Ab 0.4.3 haben saubere Farben, Kamerastabilisierung und ausreichend Platz um den Kopf Vorrang vor gleicher Gesichtsgröße. + +- HDR wird bei aktivierter Stabilisierung vor der ersten 8-Bit-Zwischendatei nach SDR umgewandelt. Stabilisierung und Export arbeiten danach in SDR; es erfolgt keine zweite HDR-Umwandlung. Ohne Stabilisierung geschieht die Umwandlung ebenfalls vor Beschleunigung und Skalierung. +- Die HDR-Umwandlung glättet Farbrauschen räumlich und zeitlich, während die Helligkeitsinformation praktisch ungeglättet bleibt. Dithering reduziert Quantisierungsstufen beim Wechsel nach 8 Bit. Das kann schwankende Farbflecken mindern, ersetzt aber keine Korrektur stark wechselnder Beleuchtung im Original. +- VidStab analysiert das ganze Bild. Der frühere ausgeschnittene Hintergrund-Anker wurde entfernt, da Analyse und Transform sonst unterschiedliche Drehzentren verwenden. Alte Ankerwerte bleiben zur Dateikompatibilität lesbar, haben aber keine Wirkung mehr. Der von VidStab berechnete Sicherheitszoom bleibt pro Clip konstant. +- Die gemeinsame Zielgröße wird nicht mehr automatisch von 26 % auf bis zu 38 % erhöht. Zusätzlicher Gesichtszoom ist auf 1,12× gegenüber dem passenden Bildformat begrenzt; notwendiger Formatbeschnitt und VidStab-Sicherheitszoom sind davon getrennt. +- Gesichtsausschnitte bleiben pro Clip fest. Sie folgen nicht mehr jeder Kopfbewegung und drehen Kopfneigung nicht gerade. Eine Reserve über und neben den erkannten Gesichtern berücksichtigt Haare und Bewegung, soweit das Ausgangsmaterial genug Rand bietet. +- Der Cache wird wegen der geänderten Verarbeitung neu aufgebaut. Ein normaler Projektrender erhält die ausgewählten Segmente. + +Die folgenden Abschnitte beschreiben die Einführung der Funktionen in 0.4.0; bei abweichendem Verhalten gelten die obigen Korrekturen. + # Visuelle Kontinuität – Version 0.4.0 Ziel ist ein ruhiger Wachstumsfilm trotz wechselndem Aufnahmeabstand, Kopfhaltung und Hintergrund. Apple Vision bleibt die Grundlage für Gesichtserkennung und Augenmerkmale; FFmpeg setzt die Bildtransformationen um. Alter und Datum werden noch nicht eingeblendet.