diff --git a/Makefile b/Makefile index 4c954248..e9fd4afd 100644 --- a/Makefile +++ b/Makefile @@ -81,6 +81,7 @@ test-contracts: @PYTHONDONTWRITEBYTECODE=1 python3 "$(ROOT)/macos/Tests/test-hvf-mapped-sections.py" @PYTHONDONTWRITEBYTECODE=1 python3 "$(ROOT)/macos/Tests/test-virtio-pinch.py" @PYTHONDONTWRITEBYTECODE=1 python3 "$(ROOT)/tests/test-build-cache.py" + @PYTHONDONTWRITEBYTECODE=1 python3 "$(ROOT)/tests/test-profile-process.py" @PYTHONDONTWRITEBYTECODE=1 python3 "$(ROOT)/tests/test-development-sign-identity.py" @PYTHONDONTWRITEBYTECODE=1 python3 "$(ROOT)/tests/test-app-version.py" @PYTHONDONTWRITEBYTECODE=1 python3 "$(ROOT)/tests/test-pack-app-icon.py" diff --git a/docs/performance.md b/docs/performance.md new file mode 100644 index 00000000..d0bd78e6 --- /dev/null +++ b/docs/performance.md @@ -0,0 +1,410 @@ +# Performance profiling + +CPU instructions execute through HVF on the Apple Silicon CPU. Graphics take a +longer path: guest Mesa → VirGL → native macOS OpenGL → Cocoa. Accelerated +graphics therefore still have command translation, synchronization, upload, and +presentation costs. See [architecture](architecture.md) and QEMU's +[virtio-gpu documentation](https://www.qemu.org/docs/master/system/devices/virtio/virtio-gpu.html). + +## Reproducible resource measurements + +`scripts/profile-process.py` reads interval CPU counters and memory accounting on +macOS and Linux. It does not launch, stop, configure, or modify a VM. Specify the +actual QEMU PID, rather than the launcher PID: + +```sh +python3 scripts/profile-process.py --pid QEMU_PID --seconds 60 --output dist/profile.json +``` + +When sampling QEMU, add `--qmp /path/to/private/qmp.sock`. The read-only QMP guard +checks the run state throughout the sample and rejects stops, resets, suspension, +shutdown, guest panic, and block-I/O errors. An invalid sample does not overwrite +an existing output file. A VM paused because its host disk is full can consume +almost no CPU; that must not be reported as healthy desktop idle. + +Record host free disk space as well as memory. Leave room for writable snapshots, +shader caches, build intermediates, and other Mac applications. Prefer APFS clones +of disposable factory images over independent copies when setting up repeated +tests. Never remove personal VM disks to make space for a benchmark. + +Repeat `--pid` to include integration helpers. CPU usage is summed across the +selected processes; memory remains separate per process to avoid silently adding +shared memory. The sampler checks process start identity and fails if a PID exits, +becomes inaccessible, or is reused. + +- `cpu_one_core_percent`: 100% means one logical CPU fully occupied. +- `cpu_host_capacity_percent`: that value divided by the host's logical CPU count. + QEMU at 10% of one core on a ten-core Mac occupies approximately 1% of total + CPU capacity. This does not account for different performance/efficiency cores. +- `system_cpu_busy_percent`: all system CPU activity, including other applications + and kernel work. Background activity cannot be attributed to QEMU. +- macOS `physical_footprint_mib`: charged memory, including compressed memory. + RSS alone can substantially understate the VM's memory cost under host pressure. +- Linux `proportional_resident_mib` and `proportional_swap_mib`: PSS and swap PSS + when permitted. These are different accounting measures from macOS footprint. + +Apple Silicon's `proc_pid_rusage` CPU times are Mach ticks, not nanoseconds. The +sampler converts using `mach_timebase_info`. A missing permission or unsupported +accounting source is an error, never a zero-usage result. + +For native Omarchy, copy the script to that machine and sample its compositor: + +```sh +python3 profile-process.py --pid "$(pgrep -x Hyprland)" --seconds 60 --output /tmp/native-profile.json +omarchy dev benchmark cli --repeat=10 +``` + +The first command also records whole-system CPU utilization, which is the useful +metric for a native OS idle comparison. Include the shell's PID to inspect its +individual memory cost. The upstream [CLI benchmark](https://github.com/basecamp/omarchy/blob/c668141e9c42b13c80c9ca4ea108e11708c5e8a5/bin/omarchy-dev-benchmark-cli) +measures common command response times; it does not measure desktop graphics. +Record CPU/GPU models, core count, OS/package versions, +power profile, resolution, scale, refresh rate, open applications, and whether +the screen is displaying the desktop, lock, or animated screensaver. + +Measure settled desktop idle separately from boot, app startup, screensaver, +video, builds, and camera use. Compare equal work and include frame-time tails, +CPU seconds per workload, and memory after applications close. Record concurrent +host activity; a quiet baseline and a loaded-host run answer different questions. +Offscreen synchronous graphics throughput does not establish on-screen FPS or +input latency. Do not infer hardware video decoding from an accelerated desktop. + +## 2026-10-01 investigation + +Host: M2 Pro, ten cores (six performance/four efficiency), 16 GiB, macOS 27.0.1. +Source: `prepare-0.5.0`, initially `3cfee41`; optimization checkpoint `0bd5348`, Bash compatibility fix `3e24c17`. +Guest: current factory pinned to Omarchy release 4.0.4 (source-reported +4.0.0.alpha), Linux 7.2.8, Hyprland 0.56.2, Aquamarine 0.15.1. + +The desktop tests used disposable snapshots of the newly provisioned factory QA +disk, not the personal VM. Cocoa used an 840×474 backing-pixel window, scale 2, +with a roughly 120 Hz EDID. Fullscreen and input grabbing were disabled while the +Mac was in use. These results do **not** certify fullscreen Retina frame pacing. +QEMU's CPU/memory figures below exclude the separate native integration helpers +and additional WindowServer/GPU costs. + +| Settled desktop, 30 seconds | QEMU CPU, one core | QEMU CPU, ten-core capacity | macOS charged footprint | Guest RAM used¹ | +| --- | ---: | ---: | ---: | ---: | +| Original renderer, 4 CPUs / 4 GiB | 9.56% | 0.96% | 2872 MiB | 729 MiB | +| Optimized renderer, 4 CPUs / 4 GiB | 9.82% | 0.98% | 2792 MiB | 729 MiB | +| Optimized renderer, default 8 CPUs / 8 GiB | 10.00% | 1.00% | 2842 MiB | 790 MiB | + +¹ `MemTotal - MemAvailable`, an estimate rather than the sum of application RSS. +The default-capacity guest also retained about 849 MiB of buffers/file cache and +had about 7.0 GiB available. Selected capacity is not physical host residency. +After repeated browser/graphics workloads the four-GiB VM's charged footprint +reached roughly 3.5 GiB; idle and workload memory should not be conflated. + +The original renderer's small-window animated screensaver used about 33% of one +host core (3.3% of total capacity). It is an active rendering workload. Earlier +QA samples near 170% of one core had no controlled workload attribution and are +not evidence of idle CPU usage. + +The user's approximately 2% native idle figure is a reference, not a measured +hardware-matched baseline. Its denominator and hardware are unknown. Neither +these results nor online reports establish a VM/native performance ratio. +The official manual's [native Mac performance example](https://learn.omacom.io/2/the-omarchy-manual/97/mac-support) +reports a gain on a 2019 Intel MacBook Pro after installing Omarchy. That is +different hardware and a different comparison; it does not measure this ARM64 VM. + +A short host stack sample showed the vCPUs predominantly waiting for interrupts, +with the main loop, ANGLE workers, and Cocoa event loop predominantly sleeping. +It did not reveal a persistent busy loop. This is a qualitative stack check, +not a statistical attribution of CPU time or a GPU utilization measurement. + +### Change and observed benefit + +VirGL was built with Meson's `debug` setting, which selects `-O0`. It now uses +`debugoptimized` (`-O2`) with `b_ndebug=false`; graphics functionality, assertions, +diagnostics, source pins, and compatibility patches are retained. All 90 captured +VirGL compile commands used `-O2` without `-DNDEBUG`. Meson's +[build-type documentation](https://mesonbuild.com/Builtin-options.html#details-for-buildtype) +describes these options. `OMARCHY_RUNTIME_BUILD_JOBS=2` also bounds compilation +while other host applications are in use. + +Alternating original/optimized runs exercised small command-heavy draws, +1280×720 fill, and 1280×720 texture uploads through the actual guest VirGL +renderer. Every run validated the returned pixel values. Warm runs overlapped: +command-heavy median frames were about 2.9 ms originally and 2.8–2.9 ms optimized; +fill about 1.2–1.5 ms and 1.4–1.7 ms; upload about 2.2–3.0 ms and 2.1–2.4 ms. +Other runs took 25–34 ms per frame under contention. The removed unoptimized path +is a concrete improvement to the build; these measurements do not establish a +reliable percentage improvement in end-to-end responsiveness. + +Factory `foot` terminal startup mapped a window in 78–254 ms before and 145–184 ms +after; Chromium with a private blank profile took 2.40–2.98 s before and +1.94–3.15 s after. These are small samples of window availability, not first +interactive frame or a controlled native comparison. Actual Wayland output was +captured and inspected after the change. + +A deterministic 50-million-iteration integer workload returned identical results +in native macOS and Linux/HVF. The last three pairs took 0.099–0.103 s on macOS and +0.102–0.108 s in HVF. Earlier VM runs took 0.179 and 0.766 s. Different compilers, +host scheduling, and core placement prevent treating this as an exact overhead +measurement; it demonstrates that CPU-bound work can execute close to native +speed, with substantial contention outliers. + +### Rejected fence-context experiment + +Stacks from an active command-heavy workload included +`virtio_gpu_fence_poll → vrend_renderer_check_fences → eglMakeCurrent → +ContextMtl::onUnMakeCurrent → Metal waitUntilScheduled`. Upstream VirGL still +forces its housekeeping context before nonthreaded fence polling. A temporary +Darwin patch instead rebound the last renderer context, retaining fence order, +nonblocking checks, and query retirement. Its source checks, compiled deterministic +tests, and runtime build passed, but that did not establish an actual GPU benefit. + +On the same current factory snapshot, the real test repeatedly created shared +and unshared GLES contexts, checked fences and occlusion-query results, validated +pixel colors, and destroyed all contexts. The unchanged optimized runtime passed +60 cycles / 2,160 frames in **8.91 seconds**; the candidate passed in **70.48 +seconds**, exceeding the original 30-second test budget. It was progressing, +not deadlocked. Basic synchronous draws completed in both versions. These are +observations under concurrent Mac use, not a controlled percentage claim, but +they reject this candidate as a useful optimization. The patch was removed; +the shipping path retains the original context and synchronization behavior. + +Late runs also encountered host disk exhaustion. QMP reported `io-error` and +the root block device reported `nospace`; CPU activity fell nearly to zero when +the VM stopped. Earlier apparent stalls and late resource figures are therefore +invalid graphics/idle evidence. Removing only the closed disposable factory copy +restored roughly 5.4 GiB; subsequent tests used an APFS clone of the current +factory image. The baseline then passed the context workload, with similar draw +times before and after it. No writable personal VM disk was removed. The sampler's +QMP guard prevents this failure from being mislabeled as efficient idle behavior. + +The published-package check found QEMU 11.1.1, Omarchy 4.0.4, ANGLE tap 1.0.16, +and VirGL tap 1.0.42 already pinned. No package upgrade was needed for these tests. +See the [QEMU download page](https://www.qemu.org/download/), +[Omarchy release](https://github.com/basecamp/omarchy/releases/tag/v4.0.4), +[ANGLE tap releases](https://github.com/startergo/homebrew-angle/releases), and +[VirGL tap releases](https://github.com/startergo/homebrew-virglrenderer/releases). + +### Memory and reliability + +The rebuilt runtime's existing disposable reclamation test verified three +768-MiB allocation/reuse cycles with a live sentinel. It returned 740.3, 754.2, +and 732.3 MiB of charged footprint. A temporary reduction of Linux's free-page +reporting order from 9 to 7 did not produce a useful saving in the desktop trial; +the default was restored and no guest memory policy was changed. + +Free-page reporting does not tell Linux to release its file cache when macOS is +under pressure. Automatically dropping caches would sacrifice useful work and +increase later I/O. A future pressure-aware balloon controller needs a guest RAM +floor, hysteresis, allocation-latency and swap-pressure measurements, failure +recovery, and long-run validation before it becomes a daily-use default. + +Settled default-capacity checks found no failed system or user services. Desktop +VMs shut down orderly between runs. No personal disk, resource preference, +clipboard, camera, or microphone content was used for these benchmarks. + +### Short repeated desktop check + +The final rebuilt runtime completed ten cycles of opening a private blank-profile +Chromium window and `foot`, validating 1920×1080 draw/upload pixels, and closing +the applications. Between cycles, guest estimated RAM used ranged from 791 to +822 MiB and ended at 807 MiB; no guest swap was used and no system/user services +failed. Warm synchronous draw medians were 1.45–1.87 ms and uploads 2.96–4.22 ms. + +A five-minute host sample overlapping this repetition and diagnostic/CLI checks +averaged 19.67% of one core (1.97% of ten-core capacity), with a median interval of +9.69% of one core. Whole-system busy CPU was 28.34%, including other Mac activity. +Charged QEMU memory ranged from 3773 to 4009 MiB and ended at 3960 MiB. These are +mixed-workload figures, not another idle comparison or proof of an all-day leak +plateau. They show why launch/working-session costs should accompany idle figures. + +The upstream CLI benchmark completed ten repetitions of each case. Plain/help +commands averaged 17.9/24.1 ms; `commands` averaged 783 ms, JSON enumeration +367/225 ms, and the remaining help/theme cases 5.4–184 ms. No native counterpart +was measured. The initial synchronous invocation exceeded the QA agent's +eight-second command limit; an asynchronous run completed with exit status zero. + +An attempted `egl-headless` launch exited before boot: this macOS runtime does +not support that display backend. The repetition used the same small Cocoa +window with input grabbing disabled. Final actual Wayland output was inspected. +The guest error journal contained a login-keyring unlock message; no workload +crash, OOM, or I/O errors appeared in that journal. That is a short smoke check, +not sleep/wake, physical audio, or long-session certification. + +### Video acceleration + +FFmpeg lists VA-API among its compiled methods, and initializing the guest's +`/dev/dri/renderD128` VA-API device succeeds. Neither proves codec acceleration. +Querying the actual VA-API 1.24 profiles returned only `VAProfileNone (-1)` with +`VAEntrypointVideoProc (10)`, with no codec decode profiles. Those constants are +defined in the [primary libva API](https://github.com/intel/libva/blob/master/va/va.h). +`vainfo` was not installed; the check called libva's profile/entrypoint APIs +directly instead. The current VirGL build also explicitly has `video=false`. + +Hardware video decoding is consequently unavailable through this VA-API path. +Adding reliable host video acceleration requires a supported guest/host bridge; +enabling a browser flag or declaring graphics accelerated does not supply one. +Video playback CPU, dropped frames, and battery cost need separate measurements. + +### Verification and remaining work + +- `OMARCHY_RUNTIME_BUILD_JOBS=2 nice -n 10 make runtime`: passed; shader/blend + regressions, libslirp tests, compatibility audits, and runtime validation passed. + An intermediate Bash fix failed because `env` cannot invoke a shell function; + the final fix keeps direct Ninja invocation with guarded array expansion. +- `nice -n 10 make app`: passed before concurrent guest-upgrade edits began; + strict app signature and all 22 Mach-O deployment targets validated. +- `OMARCHY_RUNTIME_BUILD_JOBS=2 OMARCHY_GUEST_BUILD_JOBS=2 nice -n 10 make app`: + a later current-workspace assembly rebuilt and packed the guest, including a + successful filesystem check, but the build cache rejected the result because + unrelated guest migration inputs changed during compilation. No successful + cache state was published for that attempt. +- `python3 macos/Tests/hvf-memory-reclaim-smoke.py --qemu macos/.build/qemu-gpu-runtime/bin/qemu-system-aarch64 --guest-dir dist/guest`: passed. +- `nice -n 10 make test TEST_JOBS=2`: failed with three settings-install tests + during concurrent changes to the unrelated guest-upgrade flow. Those edits + were preserved. Later workspace reruns passed, including the final Bash fix: + 383 Swift tests in 83 suites, 334 guest tests, runtime/shell contracts, and + 16 disk-resize tests. +- `nice -n 10 make -C dist/performance-2026-10-01/verification-source test TEST_JOBS=2` + against an isolated `git archive 0bd5348`: passed, including 378 Swift tests, + 310 guest tests, runtime/shell contracts, and 16 disk-resize tests. +- Sampler CPU counters agreed with real `getrusage` on macOS and Linux; malformed + durations/PIDs and invalid compilation budgets were rejected. A real Linux + container and macOS process exercised whole-system CPU sampling. Apple Bash + 3.2 checks verified unset, empty, and explicit compilation budgets, preserving + argument boundaries for paths containing spaces. `bash -n` and + `git diff --check` passed. +- `python3 tests/test-profile-process.py`: passed all four tests, including + stopped/error states, transient QMP events, malformed replies, and preserving + earlier measurements after an invalid sample. Socket binding was denied in the + sandbox; the same tests passed outside it. The updated sampler also ran against + actual Linux counters and a real QEMU QMP socket. +- After removing the fence experiment, `OMARCHY_RUNTIME_BUILD_JOBS=2 nice -n 10 make runtime` + passed again. The final `nice -n 10 make test TEST_JOBS=2` passed with the counts + above, including the profiler tests. No fence-context patch remains in the build. +- `codesign --verify --deep --strict --verbose=2 'dist/app.noindex/Try Omarchy.app'`: + the later current bundle passed outside the sandbox; the sandboxed attempt failed + with `CSSMERR_TP_NOT_TRUSTED`. `macos/verify-macos-compatibility.sh "$PWD/dist/app.noindex/Try Omarchy.app"` + passed for all 22 Mach-O images. Its compiled fence-poll function matches the + unchanged optimized baseline, rather than the archived experiment. This artifact + audit does not turn the earlier interrupted `make app` into a successful build. + +Raw logs, workload sources, JSON measurements, and guest captures are local under +`dist/performance-2026-10-01/` and are deliberately not committed. + +Priorities for further daily-use work are fullscreen 60/120-Hz frame-time and +input-latency measurements, graphics upload/synchronization costs, browser video +decoding, audio continuity under rendering load, and memory behavior during a +long browser/editor/build session alongside Mac applications. Keep existing +animations, display resolution, refresh support, checksums, durable I/O, and +recovery policies while investigating these costs. Nested virtualization on +newer chips, physical sleep/wake, long-session leak behavior, and a native +Omarchy hardware comparison were not verified by this run. + +### Live browser-lag follow-up + +On October 1 at approximately 21:12–21:16 JST, a read-only investigation of the +personal running VM followed a report of choppy browser scrolling and Brave's +setup animation, while Omarchy menus appeared smooth. The launcher selected +8 vCPUs, 8 GiB RAM, HVF, `virtio-gpu-gl-pci`, and Cocoa GLES. This establishes the +VM's accelerated graphics configuration, not the browser's acceleration status. + +A five-second `sample` of QEMU collected 2,074 samples of its `qemu_main` thread. +805 were inside `vrend_renderer_copy_transfer3d`, predominantly copying guest +pixel data and uploading it through ANGLE's Metal texture path. Fence/context +switch waits also appeared. These are sampled stacks, not measured frame times +or proof that a particular browser produced the transfers; the guest's active +workload was not controlled. + +The 16 GiB Mac reported 19,979 MiB of swap in use. A subsequent ten-second +`vm_stat` interval recorded 63.88 MiB of swap-ins, 932 MiB of compression, and +1,001.86 MiB of decompression system-wide. These counters cannot attribute +pressure to one application. A separate ten-second process sample measured +QEMU at 22.38% of one core (2.24% of ten-core host capacity), with a charged +footprint of approximately 10,185 MiB. Footprint includes compressed memory; +it is neither guest RAM usage nor evidence of a leak. + +Active host paging is therefore a credible contributor, and texture transfers +are a graphics path worth profiling under controlled scrolling. Neither finding +establishes a 15 FPS limit or distinguishes browser software rasterization from +accelerated rendering. Capture the browser's `chrome://gpu` or `brave://gpu` +feature status, renderer, and errors next, then compare frame times with host +memory pressure relieved and identical backing resolution and browser content. + +No browser flags, resource preferences, or personal VM files were changed. The +UI inspection tool could select the launcher but could not access the separate +QEMU guest window. A second VM comparison was deferred to avoid adding memory +pressure. The process sampler ran without QMP because the live app owns its +control session; this sample is not certified as healthy idle. No browser FPS, +scroll recording, or GPU-report verification was completed. Raw diagnostic +artifacts are ignored under `dist/browser-lag-2026-10-01/`. + +### Default browser acceleration fix + +The subsequent October 1 investigation reproduced the reported software-only +browser status in a disposable ARM64 guest. Chromium 153.0.8010.36 with Mesa +26.2.3 could not create its default ANGLE GLES 3 context: the host ANGLE/Metal +path exposed GLES 3.0, but VirGL exposed only desktop OpenGL 2.1 to the guest. +The GPU process logged `eglCreateContext` failures with `EGL_BAD_ATTRIBUTE`. +Smooth compositor menus did not establish browser GPU acceleration. + +The default launcher now uses native Apple OpenGL (`cocoa,gl=on`). The shared +VirGL patch preserves native multisampling and tests framebuffer formats with +mutable multisample allocation when immutable multisample storage is absent. +It also avoids reallocating multisample textures as ordinary textures, selects +integer vertex inputs automatically for Apple's native vendor, and prevents +duplicate alpha and BGRA conversions. Simply switching the display backend +without these corrections still produced browser fallback or missing text and +incorrect colors. The patch applies below all applications; no browser profile, +GPU launch flag, or system-wide vendor override is required. Existing verified +factory bundles with `angle-metal` host metadata remain compatible with the +unchanged guest virtio/VirGL interface. + +The final production runtime was exercised in a separate 4-vCPU, 2-GiB guest at +840 × 474. With a fresh default Chromium profile, its GPU report enabled canvas, +GPU compositing, rasterization, OpenGL and WebGL, identified +`ANGLE (Mesa, virgl (Apple M2 Pro), OpenGL ES 3.0 Mesa 26.2.3-arch1.1)`, and +reported four MSAA samples. A real GLES pixel probe rendered a diagonal into a +four-sample renderbuffer, resolved it, and found 16 partially covered pixels, +120 white pixels and 120 black pixels. WebGL texture probes returned the +expected RGBA and alpha values. Firefox 157 also created WebGL 2 contexts and +rendered its interface with the same vendor policy. Actual guest screenshots +and a Chromium scrolling/WebGL recording were inspected for text and color +regressions; early broken native-backend captures were rejected. + +These checks establish functional acceleration, not a universal frame-rate +guarantee. The final Chromium idle animation run measured about 48 animation +callbacks per second under host memory pressure; callbacks are not presented +frames. Startup, backing resolution, workload and host paging were not held +constant across all exploratory runs. Full-screen scrolling performance still +needs a controlled comparison. Video decode and encode remain software, and +Vulkan remains disabled. WebGPU's feature-status entry was not independently +validated as a usable device. + +Verification on macOS: + +- `OMARCHY_RUNTIME_BUILD_JOBS=2 make runtime`: passed, including the existing + renderer regressions, seven multisample format cases, native shader/vendor + regressions, and runtime signature/deployment validation. +- `python3 tests/test-build-cache.py`: passed all 12 tests. +- `macos/Tests/run-qemu-ssh-contract.test.sh`: passed, including native-default, + legacy factory compatibility and unsupported-backend rejection. +- `make test`: passed after the final code and test changes. +- `OMARCHY_GUEST_BUILD_JOBS=2 make guest`: blocked by the unrelated package lock: + the repository supplied `noto-fonts` version `1:2026.10.01-1` rather than the + pinned `1:2026.09.01-1`. Package pins were not refreshed for this graphics fix. +- `(cd dist/guest && shasum -a 256 -c SHA256SUMS)`: passed all nine existing + factory artifacts. `macos/build-app.sh --guest-dir dist/guest` passed using + that verified factory and the corrected runtime. The installed-identity + build with `--configuration production --sign-identity` and the existing + local signing certificate also passed all 22 Mach-O deployment checks. +- `OMARCHY_QEMU_GPU_INSPECT_ONLY=1 'dist/app.noindex/Try Omarchy.app/Contents/Resources/scripts/run-qemu-gpu.sh'`: + passed the bundled factory contract. `git diff --check` passed. + +Raw GPU reports, pixel probes, build logs, screenshots and recordings remain +ignored under `dist/browser-lag-2026-10-01/`. + +After the user approved restarting, the signed installed app was updated with +its existing bundle identity and a rollback copy retained in the ignored +diagnostics directory. The personal VM booted from its existing persistent disk +with `cocoa,gl=on`, 8 vCPUs and 8 GiB RAM. The installed renderer checksum matched +the signed production build, and the boot console reported a clean filesystem +and successful settings integration. The separate optional integration update +was skipped. The UI tool could not inspect the restarted QEMU guest window, so +the browser GPU report and visual checks above are from the disposable guest, +not a new measurement of the personal full-screen session. diff --git a/macos/README.md b/macos/README.md index 9d6dcc72..e5cbb98c 100644 --- a/macos/README.md +++ b/macos/README.md @@ -42,6 +42,8 @@ without shipping an unoptimized graphics command path. To leave capacity for other host applications during a runtime rebuild, use `OMARCHY_RUNTIME_BUILD_JOBS=2 make runtime`. The override bounds Ninja compilation for VirGL, libslirp, and QEMU; otherwise Ninja selects its usual parallelism. +See [performance profiling](../docs/performance.md) for CPU/memory accounting, +reproducible measurements, and the current native-comparison limitations. `make release` defaults to the maintainer's Developer ID Application identity and `try-omarchy` notarytool profile. The app builder is also directly usable @@ -136,8 +138,10 @@ the system-tool fallback. App signature and runtime validation remain unchanged. The Resources editor stores CPU count, RAM, and an optional maximum disk capacity in the versioned `vmResourcePreferences` UserDefaults value. Until the first save, it adopts the existing `memoryPreferences` choice without rewriting it. CPU choices range -from 4 through all host cores. Memory reuses `MemoryPolicy`'s 4 GiB default and -6/8/12/16 GiB choices with 8 GiB of host headroom. Saved values that no longer +from 4 through all host cores. Memory recommends 8 GiB on hosts with at least +16 GiB. Choices start at 4/6/8 GiB and continue in 4 GiB steps, keeping at least +4 GiB for macOS; larger allocations that leave less than 8 GiB carry a +performance note. Saved values that no longer fit resolve independently to their defaults without rewriting storage. The app exports `OMARCHY_QEMU_GPU_CPUS` and the established diff --git a/scripts/profile-process.py b/scripts/profile-process.py new file mode 100755 index 00000000..412af27b --- /dev/null +++ b/scripts/profile-process.py @@ -0,0 +1,280 @@ +#!/usr/bin/env python3 +"""Read-only interval CPU/memory sampling on macOS or Linux (100% = one core).""" + +import argparse +from contextlib import nullcontext +import ctypes +import datetime +import json +import math +import os +from pathlib import Path +import platform +import socket +import statistics +import sys +import time + + +class DarwinUsage(ctypes.Structure): + _fields_ = [("uuid", ctypes.c_uint8 * 16)] + [ + (name, ctypes.c_uint64) for name in ( + "user", "system", "idle_wakeups", "interrupt_wakeups", "pageins", + "wired", "resident", "footprint", "start", "exit")] + + +class Timebase(ctypes.Structure): + _fields_ = [("numer", ctypes.c_uint32), ("denom", ctypes.c_uint32)] + + +class QMPStatus: + """Observe VM state without pausing, resuming, or changing the guest.""" + + def __init__(self, path): + self.socket = socket.socket(socket.AF_UNIX) + self.socket.settimeout(3) + self.stream = None + self.sequence = 0 + try: + self.socket.connect(str(path)) + self.stream = self.socket.makefile("rwb") + if "QMP" not in self.message(): + raise RuntimeError("invalid QMP greeting") + self.command("qmp_capabilities") + except Exception: + self.close() + raise + + def message(self): + line = self.stream.readline(1024 * 1024 + 1) + if not line or len(line) > 1024 * 1024: + raise RuntimeError("QMP disconnected or exceeded the message limit") + try: + message = json.loads(line) + except (ValueError, UnicodeError) as error: + raise RuntimeError("invalid QMP message") from error + if not isinstance(message, dict): + raise RuntimeError("invalid QMP message") + return message + + def command(self, name): + self.sequence += 1 + self.stream.write((json.dumps({"execute": name, "id": self.sequence}) + "\n").encode()) + self.stream.flush() + # Bound unsolicited messages as well as individual blocking reads. + deadline = time.monotonic() + 3 + for _ in range(256): + remaining = deadline - time.monotonic() + if remaining <= 0: + raise RuntimeError("QMP reply timed out") + self.socket.settimeout(remaining) + message = self.message() + if message.get("event") in {"STOP", "RESET", "SUSPEND", "SHUTDOWN", "BLOCK_IO_ERROR", "GUEST_PANICKED"}: + raise RuntimeError(f"VM event {message['event']} invalidated this performance sample") + if message.get("id") == self.sequence: + if "error" in message or "return" not in message: + raise RuntimeError(f"QMP {name} failed: {message.get('error')}") + return message["return"] + raise RuntimeError("QMP did not reply within the message limit") + + def check(self): + status = self.command("query-status") + if not isinstance(status, dict) or status.get("running") is not True or status.get("status") != "running": + raise RuntimeError(f"VM is not running normally: {status}; low CPU here is not a desktop idle measurement") + + def close(self): + if self.stream is not None: + self.stream.close() + self.socket.close() + + def __enter__(self): + try: + self.check() + except Exception: + self.close() + raise + return self + + def __exit__(self, *args): + self.close() + + +def darwin_reader(): + libproc = ctypes.CDLL("/usr/lib/libproc.dylib", use_errno=True) + libproc.proc_pid_rusage.argtypes = [ctypes.c_int, ctypes.c_int, ctypes.c_void_p] + libproc.proc_pid_rusage.restype = ctypes.c_int + libsystem = ctypes.CDLL("/usr/lib/libSystem.B.dylib") + tb = Timebase() + if libsystem.mach_timebase_info(ctypes.byref(tb)) or not tb.denom: + raise RuntimeError("cannot read Mach timebase") + # ri_user_time/ri_system_time use Mach ticks, not nanoseconds on Apple Silicon. + seconds_per_tick = tb.numer / tb.denom / 1e9 + + def read(pid): + usage = DarwinUsage() + if libproc.proc_pid_rusage(pid, 0, ctypes.byref(usage)): + raise OSError(ctypes.get_errno(), f"cannot read PID {pid}") + return { + "identity": usage.start, + "cpu_seconds": (usage.user + usage.system) * seconds_per_tick, + "resident_mib": usage.resident / 2**20, + "physical_footprint_mib": usage.footprint / 2**20, + "package_idle_wakeups": usage.idle_wakeups, + "pageins": usage.pageins, + } + + return read + + +def linux_reader(): + ticks_per_second = os.sysconf("SC_CLK_TCK") + page_bytes = os.sysconf("SC_PAGE_SIZE") + + def read(pid): + root = Path(f"/proc/{pid}") + stat = (root / "stat").read_text() + # comm can contain spaces and parentheses. Fields after the final ')' start at 3. + fields = stat[stat.rindex(")") + 2:].split() + result = { + "identity": int(fields[19]), + "cpu_seconds": (int(fields[11]) + int(fields[12])) / ticks_per_second, + "resident_mib": int(fields[21]) * page_bytes / 2**20, + } + try: + lines = (root / "smaps_rollup").read_text().splitlines() + memory = {line.split(":")[0]: int(line.split()[1]) / 1024 + for line in lines if line.startswith(("Pss:", "SwapPss:"))} + result["proportional_resident_mib"] = memory["Pss"] + result["proportional_swap_mib"] = memory["SwapPss"] + except PermissionError: + # RSS is still available when Linux restricts detailed memory accounting. + pass + return result + + return read + + +def system_reader(system): + if system == "Linux": + def read(): + # Exclude guest/guest_nice: Linux already includes them in user/nice. + ticks = list(map(int, Path("/proc/stat").read_text().splitlines()[0].split()[1:9])) + return ticks, ticks[3] + ticks[4] + return read, None + libsystem = ctypes.CDLL("/usr/lib/libSystem.B.dylib") + libsystem.mach_host_self.restype = ctypes.c_uint32 + libsystem.host_statistics.argtypes = [ctypes.c_uint32, ctypes.c_int, + ctypes.POINTER(ctypes.c_uint32), + ctypes.POINTER(ctypes.c_uint32)] + host = libsystem.mach_host_self() + + def read(): + ticks = (ctypes.c_uint32 * 4)() + count = ctypes.c_uint32(4) + # HOST_CPU_LOAD_INFO: user, system, idle, nice; counters wrap at 32 bits. + if libsystem.host_statistics(host, 3, ticks, ctypes.byref(count)): + raise RuntimeError("cannot read host CPU counters") + return list(ticks), ticks[2] + return read, 2**32 + + +def system_percent(before, after, counter_modulus): + deltas = [end - start for start, end in zip(before[0], after[0])] + idle = after[1] - before[1] + if counter_modulus: + deltas = [delta % counter_modulus for delta in deltas] + idle %= counter_modulus + total = sum(deltas) + return 100 * (total - idle) / total if total else None + + +def positive(value): + number = float(value) + if not math.isfinite(number) or number <= 0: + raise argparse.ArgumentTypeError("must be a finite positive number") + return number + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--pid", type=int, action="append", required=True, + help="repeat to include QEMU and its integration helpers") + parser.add_argument("--seconds", type=positive, default=30) + parser.add_argument("--interval", type=positive, default=1) + parser.add_argument("--output", type=Path) + parser.add_argument("--qmp", type=Path, help="reject samples if this QEMU stops, resets, or reports an I/O error") + args = parser.parse_args() + if any(pid <= 0 for pid in args.pid) or len(set(args.pid)) != len(args.pid): + parser.error("PIDs must be positive and unique") + system = platform.system() + if system not in ("Darwin", "Linux"): + parser.error("requires macOS or Linux") + with QMPStatus(args.qmp) if args.qmp else nullcontext() as monitor: + result = collect(args, system, monitor) + text = json.dumps(result, indent=2) + "\n" + if args.output: + args.output.write_text(text) + print(text, end="") + + +def collect(args, system, monitor): + read = darwin_reader() if system == "Darwin" else linux_reader() + read_system, counter_modulus = system_reader(system) + started = time.monotonic() + first = {pid: read(pid) for pid in args.pid} + first_system = read_system() + previous, previous_time = first, started + previous_system = first_system + samples = [] + deadline = started + args.seconds + next_sample = min(deadline, started + args.interval) + while True: + time.sleep(max(0, next_sample - time.monotonic())) + if monitor is not None: + monitor.check() + current = {pid: read(pid) for pid in args.pid} + current_system = read_system() + timestamp = time.monotonic() + for pid in args.pid: + if current[pid]["identity"] != first[pid]["identity"]: + raise RuntimeError(f"PID {pid} was reused during sampling") + cpu = sum(current[pid]["cpu_seconds"] - previous[pid]["cpu_seconds"] + for pid in args.pid) + samples.append({ + "elapsed_seconds": timestamp - started, + "cpu_one_core_percent": 100 * cpu / (timestamp - previous_time), + "system_cpu_busy_percent": system_percent(previous_system, current_system, counter_modulus), + "processes": current, + }) + previous, previous_time = current, timestamp + previous_system = current_system + if timestamp >= deadline: + break + next_sample = min(deadline, next_sample + args.interval) + elapsed = previous_time - started + mean_cpu = 100 * sum(previous[pid]["cpu_seconds"] - first[pid]["cpu_seconds"] + for pid in args.pid) / elapsed + result = { + "utc": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "platform": system, + "architecture": platform.machine(), + "logical_cpu_count": os.cpu_count(), + "seconds": elapsed, + "cpu_one_core_percent": mean_cpu, + "cpu_host_capacity_percent": mean_cpu / os.cpu_count(), + "system_cpu_busy_percent": system_percent(first_system, previous_system, counter_modulus), + "median_interval_cpu_one_core_percent": statistics.median( + sample["cpu_one_core_percent"] for sample in samples), + "memory_note": "macOS footprint includes compressed memory; Linux PSS and RSS are different metrics", + "system_cpu_note": "includes every process and kernel work; cannot attribute background activity to the selected PIDs", + "qmp_state_checked": monitor is not None, + "samples": samples, + } + return result + + +if __name__ == "__main__": + try: + main() + except (OSError, RuntimeError) as error: + sys.exit(f"profile-process: {error}") diff --git a/tests/test-profile-process.py b/tests/test-profile-process.py new file mode 100644 index 00000000..39de6008 --- /dev/null +++ b/tests/test-profile-process.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""Verify that paused or failed VMs cannot produce plausible idle reports.""" +import json +import os +from pathlib import Path +import socket +import subprocess +import sys +import tempfile +import threading +import unittest + +SCRIPT = Path(__file__).resolve().parents[1] / "scripts/profile-process.py" + + +class ProfileTests(unittest.TestCase): + def run_profile(self, statuses, event=None, malformed=False): + with tempfile.TemporaryDirectory(prefix="profile-qmp-", dir="/tmp") as directory: + path = Path(directory) / "qmp.sock" + output = Path(directory) / "profile.json" + output.write_text("previous measurement\n") + server = socket.socket(socket.AF_UNIX) + try: + server.bind(str(path)) + server.listen(1) + server.settimeout(5) + except Exception: + server.close() + raise + failures = [] + + def respond(): + try: + with server.accept()[0] as client, client.makefile("rwb") as stream: + stream.write(b'{"QMP":{"version":{},"capabilities":[]}}\n') + stream.flush() + checks = 0 + for line in stream: + request = json.loads(line) + response = {} + if request["execute"] == "query-status": + if malformed: + stream.write(b"invalid JSON\n") + stream.flush() + break + if event and checks == 1: + stream.write((json.dumps({"event": event}) + "\n").encode()) + state = statuses[min(checks, len(statuses) - 1)] + response = {"status": state, "running": state == "running"} + checks += 1 + else: + assert request["execute"] == "qmp_capabilities" + stream.write((json.dumps({"return": response, "id": request["id"]}) + "\n").encode()) + stream.flush() + except (BrokenPipeError, ConnectionResetError): + pass # The collector intentionally closes after detecting an invalid sample. + except Exception as error: + failures.append(error) + + thread = threading.Thread(target=respond, daemon=True) + thread.start() + try: + result = subprocess.run([ + sys.executable, str(SCRIPT), "--pid", str(os.getpid()), + "--seconds", ".08", "--interval", ".02", "--qmp", str(path), + "--output", str(output), + ], capture_output=True, text=True, timeout=10) + thread.join(5) + self.assertFalse(thread.is_alive()) + self.assertEqual(failures, []) + return result, output.read_text() + finally: + server.close() + + def test_running_vm(self): + result, output = self.run_profile(["running"]) + self.assertEqual(result.returncode, 0, result.stderr) + profile = json.loads(output) + self.assertTrue(profile["qmp_state_checked"]) + self.assertGreaterEqual(len(profile["samples"]), 1) + self.assertEqual(json.loads(result.stdout), profile) + + def test_stopped_vm_does_not_overwrite_previous_measurement(self): + for states in (["io-error"], ["running", "paused"], ["running", "io-error"]): + with self.subTest(states=states): + result, output = self.run_profile(states) + self.assertNotEqual(result.returncode, 0) + self.assertIn("not running normally", result.stderr) + self.assertEqual(output, "previous measurement\n") + + def test_transient_vm_stop_is_not_hidden_by_later_running_status(self): + for event in ("STOP", "BLOCK_IO_ERROR", "RESET", "SUSPEND"): + with self.subTest(event=event): + result, output = self.run_profile(["running"], event=event) + self.assertNotEqual(result.returncode, 0) + self.assertIn(f"VM event {event}", result.stderr) + self.assertEqual(output, "previous measurement\n") + + def test_malformed_qmp(self): + result, output = self.run_profile(["running"], malformed=True) + self.assertNotEqual(result.returncode, 0) + self.assertIn("invalid QMP message", result.stderr) + self.assertEqual(output, "previous measurement\n") + + +if __name__ == "__main__": + unittest.main()