diff --git a/kernel/src/Api/BuildNo.hpp b/kernel/src/Api/BuildNo.hpp index 34ee96d..43b7931 100644 --- a/kernel/src/Api/BuildNo.hpp +++ b/kernel/src/Api/BuildNo.hpp @@ -12,4 +12,4 @@ #pragma once -#define MONTAUK_BUILD_NUMBER 4 +#define MONTAUK_BUILD_NUMBER 6 diff --git a/kernel/src/Api/Graphics.hpp b/kernel/src/Api/Graphics.hpp index cb6704c..7d012b9 100644 --- a/kernel/src/Api/Graphics.hpp +++ b/kernel/src/Api/Graphics.hpp @@ -10,6 +10,7 @@ #include #include #include +#include #include #include "Syscall.hpp" @@ -59,9 +60,42 @@ namespace montauk::abi { } } + // When page flipping is available, map the second scanout buffer + // immediately after the first so a double-buffered client can reach + // both. Single-buffer clients never touch the extra pages. + if (Drivers::Graphics::IntelGPU::FlipSupported()) { + uint64_t buf1Phys = Drivers::Graphics::IntelGPU::GetBufferPhys(1); + for (uint64_t i = 0; i < numPages; i++) { + if (!Memory::VMM::Paging::MapUserInWC( + proc->pml4Phys, + buf1Phys + i * 0x1000, + userVa + (numPages + i) * 0x1000 + )) { + return 0; + } + } + } + return userVa; } + // SYS_FBFLIP: point scanout at buffer `index` (0 or 1). The hardware + // latches the new surface at vblank, so the flip is tear-free. + // index == -1 queries support without side effects. + static int64_t Sys_FbFlip(uint64_t index, uint64_t flags) { + namespace GPU = Drivers::Graphics::IntelGPU; + + if ((int64_t)index == -1) { + return GPU::FlipSupported() ? 1 : 0; + } + if (!GPU::FlipSupported()) return -1; + + auto* proc = Sched::GetCurrentProcessPtr(); + if (proc != nullptr) GPU::SetFlipOwner(proc->pid); + + return GPU::Flip((int)index, (flags & 1) != 0); + } + static uint64_t Sys_TermSize() { // If the process is redirected to a GUI terminal, return those dimensions auto* proc = Sched::GetCurrentProcessPtr(); diff --git a/kernel/src/Api/Syscall.cpp b/kernel/src/Api/Syscall.cpp index a731c2f..0b54a95 100644 --- a/kernel/src/Api/Syscall.cpp +++ b/kernel/src/Api/Syscall.cpp @@ -148,6 +148,8 @@ namespace montauk::abi { return 0; case SYS_FBMAP: return (int64_t)Sys_FbMap(); + case SYS_FBFLIP: + return Sys_FbFlip(frame->arg1, frame->arg2); case SYS_TERMSIZE: return (int64_t)Sys_TermSize(); case SYS_GETARGS: diff --git a/kernel/src/Api/Syscall.hpp b/kernel/src/Api/Syscall.hpp index 703e93b..670ae28 100644 --- a/kernel/src/Api/Syscall.hpp +++ b/kernel/src/Api/Syscall.hpp @@ -279,6 +279,9 @@ namespace montauk::abi { /* Power.hpp -- CPU power/thermal status */ static constexpr uint64_t SYS_POWERINFO = 149; // (PowerInfo*) -> 0, -1 unsupported + /* Graphics.hpp -- framebuffer page flip (double-buffered scanout) */ + static constexpr uint64_t SYS_FBFLIP = 150; // (index, flags) -> new front index; index=-1 queries support (1/0); flags bit0 = wait vsync + // Tunable parameters (for SYS_SDR_SETPARAM / SYS_SDR_GETPARAM). static constexpr int SDR_PARAM_FREQ = 0; // center frequency, Hz static constexpr int SDR_PARAM_SAMPLE_RATE = 1; // sample rate, Hz diff --git a/kernel/src/Common/Panic.cpp b/kernel/src/Common/Panic.cpp index 2d80852..bee58a4 100644 --- a/kernel/src/Common/Panic.cpp +++ b/kernel/src/Common/Panic.cpp @@ -5,6 +5,7 @@ #include "Panic.hpp" #include "../CppLib/BoxUI.hpp" +#include "../Drivers/Graphics/IntelGPU.hpp" static constexpr int BoxWidth = 72; @@ -65,6 +66,11 @@ static void PrintRegisters(System::PanicFrame* frame) { } void Panic(const char *meditationString, System::PanicFrame* frame) { + // Bring scanout back to buffer 0 first: if a page-flipped client was + // showing the other buffer, the panic box would land on an invisible + // surface. Safe no-op when the GPU driver is inactive. + Drivers::Graphics::IntelGPU::PanicRestoreScanout(); + // Header kerr << BOXUI_ANSI_RED_BG << BOXUI_ANSI_WHITE_FG << BOXUI_ANSI_BOLD << "\n"; PrintHorizontalEdge(BOXUI_TL, BOXUI_TR); diff --git a/kernel/src/Drivers/Graphics/IntelGPU.cpp b/kernel/src/Drivers/Graphics/IntelGPU.cpp index a0da217..f07ef1d 100644 --- a/kernel/src/Drivers/Graphics/IntelGPU.cpp +++ b/kernel/src/Drivers/Graphics/IntelGPU.cpp @@ -15,6 +15,9 @@ #include #include #include +#include +#include +#include using namespace Kt; @@ -45,6 +48,19 @@ namespace Drivers::Graphics::IntelGPU { static uint64_t g_fbSize = 0; // Total framebuffer size in bytes static uint64_t g_fbGttOffset = 0; // GTT offset where FB starts (in bytes) + // Page flip state + static volatile void* g_ggtt = nullptr; // GGTT at the gen-correct BAR0 offset + static uint64_t g_ggttEntries = 0; + static uint64_t g_fbGttOffsetA = 0; // GGTT offset of buffer 0 (firmware FB) + static uint32_t* g_buf1Virt = nullptr; // buffer 1 kernel VA (HHDM, WC) + static uint64_t g_buf1Phys = 0; + static uint64_t g_buf1GttOffset = 0; // GGTT offset of buffer 1 (in bytes) + static bool g_flipSupported = false; + static int g_frontBuffer = 0; + static int g_flipOwnerPid = -1; + static volatile uint64_t g_vblankCount = 0; + static bool g_vblankIrqReady = false; + // ========================================================================= // Register access helpers // ========================================================================= @@ -481,6 +497,284 @@ namespace Drivers::Graphics::IntelGPU { << ", stride=" << base::dec << g_fbPitch; } + // ========================================================================= + // Page flipping: GGTT at the generation-correct offset + // ========================================================================= + // + // The GTT window lives in the upper half of GTTMMADR (BAR0): + // Gen 6-7: 4 MB BAR, GTT at +2 MB, 32-bit entries + // Gen 8+: 16 MB BAR, GGTT at +8 MB, 64-bit entries + // The legacy init path above writes at +2 MB for all generations; on + // Gen 8+ those writes land in register space, and scanout keeps working + // only because the firmware's real GGTT entries (stored in stolen RAM, + // which survives S3 in self-refresh) are never actually touched. Page + // flipping needs entries the display engine really reads, so it uses the + // correct window and refuses to run unless the firmware's own scanout + // PTEs are visible there (valid bits set). On any failure page flipping + // stays off and the driver behaves exactly as before. + + static void MicroDelay(int us) { + // Simple busy-wait; us is approximate + for (volatile int i = 0; i < us * 100; i++) { + asm volatile("pause"); + } + } + + static uint64_t ReadGgttPte(uint64_t index) { + if (g_gpuGen >= 8) return ((volatile uint64_t*)g_ggtt)[index]; + return ((volatile uint32_t*)g_ggtt)[index]; + } + + static void WriteGgttPte(uint64_t index, uint64_t physAddr) { + if (g_gpuGen >= 8) { + ((volatile uint64_t*)g_ggtt)[index] = MakeGttPte64(physAddr); + } else { + ((volatile uint32_t*)g_ggtt)[index] = MakeGttPte32(physAddr); + } + } + + static bool MapCorrectGgtt() { + if (g_gpuGen < 8) { + // The legacy mapping at BAR0 + 2 MB is already the real GTT + g_ggtt = g_gttBase; + g_ggttEntries = g_gttEntryCount; + return g_ggtt != nullptr; + } + + // Gen 8+: GGC.GGMS moved to bits 7:6 (0 = none, 1/2/3 = 2/4/8 MB) + uint16_t gmchCtl = Pci::LegacyRead16(g_gpuInfo.pciBus, g_gpuInfo.pciDevice, + g_gpuInfo.pciFunction, (uint8_t)PCI_REG_GMCH_CTL); + uint8_t ggms = (gmchCtl >> 6) & 0x3; + if (ggms == 0) { + KernelLogStream(WARNING, "IntelGPU") << "GGC reports no GGTT (GGC=" + << base::hex << (uint64_t)gmchCtl << "), page flip unavailable"; + return false; + } + uint64_t ggttBytes = (1ull << ggms) * 1024 * 1024; + uint64_t ggttPhys = g_gpuInfo.mmioPhys + 8 * 1024 * 1024; + + for (uint64_t off = 0; off < ggttBytes; off += 0x1000) { + Memory::VMM::g_paging->MapMMIO(ggttPhys + off, Memory::HHDM(ggttPhys + off)); + } + + g_ggtt = (volatile void*)Memory::HHDM(ggttPhys); + g_ggttEntries = ggttBytes / sizeof(uint64_t); + + KernelLogStream(INFO, "IntelGPU") << "GGTT window at BAR0+8MB (phys " + << base::hex << ggttPhys << "), " << base::dec << (ggttBytes / 1024) + << " KB, " << g_ggttEntries << " entries"; + return true; + } + + static bool VerifyFirmwareScanoutPtes() { + // The active scanout surface must be backed by valid PTEs in the + // window we mapped; if not, our idea of where the GGTT lives is wrong + // and writing to it could corrupt arbitrary state. + g_fbGttOffsetA = ReadReg(DSPASURF) & ~0xFFFull; + uint64_t firstIdx = g_fbGttOffsetA >> 12; + uint64_t lastIdx = (g_fbGttOffsetA + g_fbSize - 1) >> 12; + + if (lastIdx >= g_ggttEntries) { + KernelLogStream(WARNING, "IntelGPU") << "Scanout range exceeds GGTT (" + << base::dec << lastIdx << " >= " << g_ggttEntries << ")"; + return false; + } + + uint64_t checks[3] = { firstIdx, (firstIdx + lastIdx) / 2, lastIdx }; + for (int i = 0; i < 3; i++) { + uint64_t pte = ReadGgttPte(checks[i]); + if (!(pte & 1)) { + KernelLogStream(WARNING, "IntelGPU") << "Scanout PTE " << base::dec + << checks[i] << " invalid (" << base::hex << pte + << ") - GGTT window not where expected, page flip unavailable"; + return false; + } + } + + KernelLogStream(OK, "IntelGPU") << "Firmware scanout PTEs verified (surface at GGTT+" + << base::hex << g_fbGttOffsetA << ", PTE[" << base::dec << firstIdx + << "]=" << base::hex << ReadGgttPte(firstIdx) << ")"; + return true; + } + + static bool AllocateBackBuffer() { + uint64_t pages = (g_fbSize + 0xFFF) >> 12; + void* virt = Memory::g_pfa->ReallocConsecutive(nullptr, pages); + if (!virt) { + KernelLogStream(WARNING, "IntelGPU") << "Failed to allocate " << base::dec + << pages << " contiguous pages for the second scanout buffer"; + return false; + } + g_buf1Phys = Memory::SubHHDM(virt); + g_buf1Virt = (uint32_t*)virt; + + // Display scanout does not snoop the CPU cache: remap the buffer + // write-combining before touching it, same as the firmware FB. + for (uint64_t i = 0; i < pages; i++) { + Memory::VMM::g_paging->MapWC(g_buf1Phys + i * 0x1000, + Memory::HHDM(g_buf1Phys + i * 0x1000)); + } + Memory::VMM::FlushTLB(); + + memset(virt, 0, pages * 0x1000); + return true; + } + + static bool FindAndMapGgttRange() { + uint64_t pages = (g_fbSize + 0xFFF) >> 12; + constexpr uint64_t guard = 16; + uint64_t need = pages + 2 * guard; + + // Firmware does not leave unused GGTT entries invalid: it points the + // ENTIRE table at a scratch page, so free entries are valid but all + // hold one identical PTE value (HW-confirmed on Raptor Lake, where a + // scan for invalid entries found none in 1M entries). Sample the tail + // of the table to learn the scratch value; a non-uniform tail means + // the layout is not what we expect, so bail. + uint64_t scratchPte = ReadGgttPte(g_ggttEntries - 1); + for (uint64_t probe = 2; probe <= 32; probe++) { + if (ReadGgttPte(g_ggttEntries - probe) != scratchPte) { + KernelLogStream(WARNING, "IntelGPU") << "GGTT tail not uniform (entry -" + << base::dec << probe << " != " << base::hex << scratchPte + << "), page flip unavailable"; + return false; + } + } + + // Search the upper half for a run of scratch-backed (or invalid) + // entries. Never overlap the active scanout range. + uint64_t avoidFirst = g_fbGttOffsetA >> 12; + uint64_t avoidLast = (g_fbGttOffsetA + g_fbSize - 1) >> 12; + + uint64_t runStart = 0, run = 0; + bool found = false; + for (uint64_t idx = g_ggttEntries / 2; idx < g_ggttEntries; idx++) { + if (idx >= avoidFirst && idx <= avoidLast) { run = 0; continue; } + uint64_t pte = ReadGgttPte(idx); + if ((pte & 1) && pte != scratchPte) { run = 0; continue; } + if (run == 0) runStart = idx; + if (++run >= need) { found = true; break; } + } + if (!found) { + KernelLogStream(WARNING, "IntelGPU") << "No free GGTT run of " << base::dec + << need << " entries in upper half, page flip unavailable"; + return false; + } + + uint64_t base = runStart + guard; + for (uint64_t i = 0; i < pages; i++) { + WriteGgttPte(base + i, g_buf1Phys + i * 0x1000); + } + + // Read back first and last entry to confirm the writes stuck + uint64_t expectFirst = (g_gpuGen >= 8) ? MakeGttPte64(g_buf1Phys) + : MakeGttPte32(g_buf1Phys); + uint64_t lastPhys = g_buf1Phys + (pages - 1) * 0x1000; + uint64_t expectLast = (g_gpuGen >= 8) ? MakeGttPte64(lastPhys) + : MakeGttPte32(lastPhys); + if (ReadGgttPte(base) != expectFirst || ReadGgttPte(base + pages - 1) != expectLast) { + KernelLogStream(WARNING, "IntelGPU") << "GGTT read-back mismatch at entry " + << base::dec << base << ", page flip unavailable"; + return false; + } + + g_buf1GttOffset = base << 12; + KernelLogStream(OK, "IntelGPU") << "Buffer 1 mapped at GGTT+" << base::hex + << g_buf1GttOffset << " (" << base::dec << pages << " pages, phys " + << base::hex << g_buf1Phys << ")"; + return true; + } + + // ========================================================================= + // Page flipping: vblank interrupt (Gen 11+) + // ========================================================================= + + static void HandleInterrupt(uint8_t /*irq*/) { + // Gen 11+ flow: park the master enable, ack the pipe IIR, re-enable + uint32_t master = ReadReg(GFX_MSTR_INTR); + if (!(master & GFX_MSTR_INTR_DISPLAY)) { + WriteReg(GFX_MSTR_INTR, GFX_MSTR_INTR_ENABLE); + return; + } + WriteReg(GFX_MSTR_INTR, 0); + + uint32_t iir = ReadReg(DE_PIPE_A_IIR); + if (iir) WriteReg(DE_PIPE_A_IIR, iir); // write-1-to-clear + + WriteReg(GFX_MSTR_INTR, GFX_MSTR_INTR_ENABLE); + + if (iir & DE_PIPE_VBLANK) { + g_vblankCount = g_vblankCount + 1; + Sched::WakeObjectWaiters((void*)&g_vblankCount); + } + } + + static bool SetupMsi() { + uint8_t bus = g_gpuInfo.pciBus; + uint8_t dev = g_gpuInfo.pciDevice; + uint8_t func = g_gpuInfo.pciFunction; + + uint8_t cap = Pci::FindCapability(bus, dev, func, Pci::PCI_CAP_MSI); + if (cap == 0) { + KernelLogStream(INFO, "IntelGPU") << "MSI capability not found"; + return false; + } + + uint16_t msgCtrl = Pci::LegacyRead16(bus, dev, func, cap + 2); + bool is64bit = (msgCtrl & (1 << 7)) != 0; + + Pci::LegacyWrite32(bus, dev, func, cap + 4, MSI_ADDR_BASE); + if (is64bit) { + Pci::LegacyWrite32(bus, dev, func, cap + 8, 0); + Pci::LegacyWrite16(bus, dev, func, cap + 12, MSI_VECTOR); + } else { + Pci::LegacyWrite16(bus, dev, func, cap + 8, MSI_VECTOR); + } + + msgCtrl &= ~(0x70); // one message + msgCtrl |= (1 << 0); // MSI enable + Pci::LegacyWrite16(bus, dev, func, cap + 2, msgCtrl); + + uint16_t pciCmd = Pci::LegacyRead16(bus, dev, func, (uint8_t)Pci::PCI_REG_COMMAND); + pciCmd |= Pci::PCI_CMD_INTX_DISABLE; + Pci::LegacyWrite16(bus, dev, func, (uint8_t)Pci::PCI_REG_COMMAND, pciCmd); + + Hal::RegisterIrqHandler(MSI_IRQ, HandleInterrupt); + + KernelLogStream(OK, "IntelGPU") << "MSI enabled: vector " << base::dec + << (uint64_t)MSI_VECTOR << " (IRQ slot " << (uint64_t)MSI_IRQ << ")" + << (is64bit ? " [64-bit]" : " [32-bit]"); + return true; + } + + static void EnableVblankIrq() { + WriteReg(DE_PIPE_A_IIR, 0xFFFFFFFFu); // clear stale events + WriteReg(DE_PIPE_A_IMR, ~DE_PIPE_VBLANK); // unmask vblank only + WriteReg(DE_PIPE_A_IER, DE_PIPE_VBLANK); + WriteReg(DISPLAY_INT_CTL, DISPLAY_IRQ_ENABLE); + WriteReg(GFX_MSTR_INTR, GFX_MSTR_INTR_ENABLE); + (void)ReadReg(GFX_MSTR_INTR); + } + + static void SetupPageFlip() { + if (!MapCorrectGgtt()) return; + if (!VerifyFirmwareScanoutPtes()) return; + if (!AllocateBackBuffer()) return; + if (!FindAndMapGgttRange()) return; + + // Vblank interrupts: Gen 11+ register layout only. Older generations + // still flip, with DSPASURFLIVE polling standing in for the IRQ. + if (g_gpuGen >= 11 && SetupMsi()) { + EnableVblankIrq(); + g_vblankIrqReady = true; + } + + g_flipSupported = true; + g_frontBuffer = 0; + KernelLogStream(OK, "IntelGPU") << "Page flip ready: 2 buffers, vblank IRQ " + << (g_vblankIrqReady ? "on" : "off (SURFLIVE polling)"); + } + // ========================================================================= // Public API // ========================================================================= @@ -570,6 +864,8 @@ namespace Drivers::Graphics::IntelGPU { ProgramDisplayPlane(); g_initialized = true; + SetupPageFlip(); + uint64_t fwWidth = ::Graphics::Framebuffer::GetWidth(); uint64_t fwHeight = ::Graphics::Framebuffer::GetHeight(); uint64_t fwPitch = ::Graphics::Framebuffer::GetPitch(); @@ -680,6 +976,78 @@ namespace Drivers::Graphics::IntelGPU { return g_fbPitch; } + // ========================================================================= + // Page flip public API + // ========================================================================= + + bool FlipSupported() { + return g_flipSupported; + } + + uint64_t GetBufferPhys(int index) { + if (index == 0) return g_fbPhysBase; + if (index == 1 && g_flipSupported) return g_buf1Phys; + return 0; + } + + int GetFrontBuffer() { + return g_frontBuffer; + } + + uint64_t GetVblankCount() { + return g_vblankCount; + } + + bool WaitVblank(uint64_t timeoutMs) { + if (!g_vblankIrqReady) return false; + uint64_t start = g_vblankCount; + // The vblank IRQ fires every frame, so a wakeup lost to the race + // between reading the counter and blocking only costs one frame; + // the timeout is a backstop for a stalled pipe. + Sched::BlockOnObject((void*)&g_vblankCount, timeoutMs ? timeoutMs : 50); + return g_vblankCount != start; + } + + int64_t Flip(int index, bool waitVsync) { + if (!g_flipSupported || index < 0 || index > 1) return -1; + + uint64_t gttOff = (index == 0) ? g_fbGttOffsetA : g_buf1GttOffset; + WriteReg(DSPASURF, (uint32_t)gttOff); + (void)ReadReg(DSPASURF); + g_frontBuffer = index; + + if (waitVsync) { + if (g_vblankIrqReady) { + WaitVblank(50); + } + // Confirm the latch: DSPASURFLIVE tracks the surface actually + // being scanned out. With a working IRQ this succeeds on the + // first read; without one it is the vsync wait itself. + for (int i = 0; i < 3000; i++) { + if ((ReadReg(DSPASURFLIVE) & ~0xFFFu) == (uint32_t)gttOff) break; + MicroDelay(10); + } + } + return index; + } + + void SetFlipOwner(int pid) { + g_flipOwnerPid = pid; + } + + void OnProcessExit(int pid) { + if (!g_flipSupported || pid != g_flipOwnerPid) return; + g_flipOwnerPid = -1; + if (g_frontBuffer != 0) Flip(0, false); + } + + void PanicRestoreScanout() { + if (!g_flipSupported || !g_mmioBase || g_frontBuffer == 0) return; + WriteReg(DSPASURF, (uint32_t)g_fbGttOffsetA); + (void)ReadReg(DSPASURF); + g_frontBuffer = 0; + } + void Reinitialize() { if (!g_initialized || !g_mmioBase) return; @@ -760,6 +1128,31 @@ namespace Drivers::Graphics::IntelGPU { // 5. Reprogram display plane to point at our GTT-mapped framebuffer ProgramDisplayPlane(); + // 6. Restore page-flip state: rewrite buffer 1's GGTT entries, bring + // MSI + vblank interrupts back (the PCI config reset cleared MSI), + // and re-show whichever buffer was front when we suspended. + if (g_flipSupported) { + uint64_t pages = (g_fbSize + 0xFFF) >> 12; + uint64_t baseIdx = g_buf1GttOffset >> 12; + for (uint64_t i = 0; i < pages; i++) { + WriteGgttPte(baseIdx + i, g_buf1Phys + i * 0x1000); + } + (void)ReadGgttPte(baseIdx + pages - 1); + + if (g_vblankIrqReady) { + SetupMsi(); + EnableVblankIrq(); + } + + if (g_frontBuffer != 0) { + WriteReg(DSPASURF, (uint32_t)g_buf1GttOffset); + (void)ReadReg(DSPASURF); + } + + KernelLogStream(DEBUG, "IntelGPU") << "Page flip state restored (front=" + << base::dec << (uint64_t)g_frontBuffer << ")"; + } + KernelLogStream(OK, "IntelGPU") << "Display restored after S3 resume"; } diff --git a/kernel/src/Drivers/Graphics/IntelGPU.hpp b/kernel/src/Drivers/Graphics/IntelGPU.hpp index 4ae0fbe..1a21738 100644 --- a/kernel/src/Drivers/Graphics/IntelGPU.hpp +++ b/kernel/src/Drivers/Graphics/IntelGPU.hpp @@ -226,6 +226,34 @@ namespace Drivers::Graphics::IntelGPU { static constexpr uint32_t CURABASE = 0x70084; static constexpr uint32_t CURAPOS = 0x70088; + // --- Live surface address (read-only, reflects the surface the display + // engine is currently scanning out; DSPASURF writes latch at vblank) --- + static constexpr uint32_t DSPASURFLIVE = 0x701AC; + + // --- Display engine interrupts (Gen 8+ pipe block, Gen 11+ master) --- + // Per-pipe interrupt registers, pipe A instance. Vblank is bit 0. + static constexpr uint32_t DE_PIPE_A_ISR = 0x44400; + static constexpr uint32_t DE_PIPE_A_IMR = 0x44404; + static constexpr uint32_t DE_PIPE_A_IIR = 0x44408; + static constexpr uint32_t DE_PIPE_A_IER = 0x4440C; + static constexpr uint32_t DE_PIPE_VBLANK = (1u << 0); + + // Gen 11+ display interrupt master control (0x44200 was the Gen 8-10 + // master IRQ register; on Gen 11+ it controls the display half only) + static constexpr uint32_t DISPLAY_INT_CTL = 0x44200; + static constexpr uint32_t DISPLAY_IRQ_ENABLE = (1u << 31); + static constexpr uint32_t DISPLAY_IRQ_PIPE_A = (1u << 16); + + // Gen 11+ top-level graphics master interrupt + static constexpr uint32_t GFX_MSTR_INTR = 0x190010; + static constexpr uint32_t GFX_MSTR_INTR_ENABLE = (1u << 31); + static constexpr uint32_t GFX_MSTR_INTR_DISPLAY = (1u << 16); + + // --- MSI (vblank interrupt delivery) --- + static constexpr uint8_t MSI_IRQ = 28; // IRQ slot 28 = vector 60 + static constexpr uint32_t MSI_VECTOR = 60; + static constexpr uint32_t MSI_ADDR_BASE = 0xFEE00000; + // --- Output connectors --- static constexpr uint32_t ADPA = 0x61100; // Analog Display Port (VGA/CRT) static constexpr uint32_t DVOB = 0x61140; // DVO-B @@ -434,4 +462,40 @@ namespace Drivers::Graphics::IntelGPU { uint64_t GetHeight(); uint64_t GetPitch(); + // ========================================================================= + // Page flipping (double-buffered scanout) + // ========================================================================= + + // True when a second scanout buffer exists and the display plane can be + // flipped between the two. Requires a validated GGTT mapping. + bool FlipSupported(); + + // Physical base of scanout buffer 0 (firmware FB) or 1 (kernel-allocated) + uint64_t GetBufferPhys(int index); + + // Index of the buffer currently programmed for scanout + int GetFrontBuffer(); + + // Point the display plane at the given buffer. The hardware latches the + // new surface address at the next vblank (tear-free). When waitVsync is + // set, blocks until the flip has been latched. Returns the new front + // buffer index, or -1 if flipping is unavailable. + int64_t Flip(int index, bool waitVsync); + + // Block until the next vblank (Gen 11+ with working IRQ only). + // Returns false on timeout or when no vblank interrupt is available. + bool WaitVblank(uint64_t timeoutMs); + + // Monotonic vblank counter (0 when no vblank IRQ) + uint64_t GetVblankCount(); + + // Track which process performs flips so scanout can be restored to + // buffer 0 when it exits (called from the flip syscall / scheduler). + void SetFlipOwner(int pid); + void OnProcessExit(int pid); + + // Panic-safe: force scanout back to buffer 0 so panic output is visible. + // No locks, no allocation, no blocking. + void PanicRestoreScanout(); + }; diff --git a/kernel/src/Sched/Scheduler.cpp b/kernel/src/Sched/Scheduler.cpp index d8d3361..6a110b1 100644 --- a/kernel/src/Sched/Scheduler.cpp +++ b/kernel/src/Sched/Scheduler.cpp @@ -25,6 +25,7 @@ #include #include #include +#include #include // Assembly: context switch with CR3 and FPU state parameters @@ -1110,6 +1111,11 @@ namespace Sched { // process was not the owner). Drivers::USB::Bluetooth::A2dp::ReleaseOutput(exitingPid); + // Restore scanout to buffer 0 if the exiting process owned page + // flips, so the next fullscreen client and the kernel terminal are + // never stranded on the invisible buffer (no-op for non-owners). + Drivers::Graphics::IntelGPU::OnProcessExit(exitingPid); + // Release process-scoped IPC handles/mappings before tearing down the address space. Ipc::CleanupProcessSlot(slot, exitingPid, proc.pml4Phys); montauk::abi::CleanupHeapForSlot(slot, proc.pml4Phys); diff --git a/programs/include/Api/Syscall.hpp b/programs/include/Api/Syscall.hpp index c6f31b6..69eebb6 100644 --- a/programs/include/Api/Syscall.hpp +++ b/programs/include/Api/Syscall.hpp @@ -203,6 +203,9 @@ namespace montauk::abi { // CPU power/thermal status static constexpr uint64_t SYS_POWERINFO = 149; // (PowerInfo*) -> 0, -1 unsupported + // Framebuffer page flip (double-buffered scanout) + static constexpr uint64_t SYS_FBFLIP = 150; // (index, flags) -> new front index; index=-1 queries support (1/0); flags bit0 = wait vsync + // Tunable parameters (for SYS_SDR_SETPARAM / SYS_SDR_GETPARAM). static constexpr int SDR_PARAM_FREQ = 0; // center frequency, Hz static constexpr int SDR_PARAM_SAMPLE_RATE = 1; // sample rate, Hz diff --git a/programs/include/gui/framebuffer.hpp b/programs/include/gui/framebuffer.hpp index 303d657..68bd7bc 100644 --- a/programs/include/gui/framebuffer.hpp +++ b/programs/include/gui/framebuffer.hpp @@ -14,10 +14,12 @@ namespace gui { class Framebuffer { uint32_t* hw_fb; + uint32_t* hw_fb2; // second scanout buffer (page flip), nullptr if unsupported uint32_t* back_buf; int fb_width; int fb_height; int fb_pitch; // in bytes + int hw_next; // scanout buffer the next flip() will present (page flip only) static inline void fill_pixels(uint32_t* dst, int count, uint32_t pixel) { if (!dst || count <= 0) return; @@ -60,7 +62,8 @@ class Framebuffer { } public: - Framebuffer() : hw_fb(nullptr), back_buf(nullptr), fb_width(0), fb_height(0), fb_pitch(0) { + Framebuffer() : hw_fb(nullptr), hw_fb2(nullptr), back_buf(nullptr), + fb_width(0), fb_height(0), fb_pitch(0), hw_next(1) { montauk::abi::FbInfo info; montauk::fb_info(&info); @@ -70,6 +73,13 @@ public: hw_fb = (uint32_t*)montauk::fb_map(); back_buf = (uint32_t*)montauk::alloc((uint64_t)fb_height * fb_pitch); + + // Hardware page flipping: fb_map() maps the second scanout buffer + // directly after the first when the kernel supports flipping. + if (hw_fb && montauk::fb_flip(-1, 0) == 1) { + uint64_t pages = ((uint64_t)fb_height * fb_pitch + 0xFFF) / 0x1000; + hw_fb2 = (uint32_t*)((uint8_t*)hw_fb + pages * 0x1000); + } } int width() const { return fb_width; } @@ -250,13 +260,28 @@ public: inline void flip() { if (!hw_fb || !back_buf) return; + // With hardware page flipping, copy into the scanout buffer that is + // currently OFF screen, then ask the display engine to present it at + // the next vblank (tear-free). Without it, fall back to copying + // straight into the visible framebuffer. + uint32_t* dst_fb = hw_fb; + if (hw_fb2) dst_fb = (hw_next == 1) ? hw_fb2 : hw_fb; + // Copy back buffer to hardware framebuffer, row by row (pitch may differ) uint64_t row_bytes = (uint64_t)fb_width * sizeof(uint32_t); for (int y = 0; y < fb_height; y++) { uint32_t* src = (uint32_t*)((uint8_t*)back_buf + y * fb_pitch); - uint32_t* dst = (uint32_t*)((uint8_t*)hw_fb + y * fb_pitch); + uint32_t* dst = (uint32_t*)((uint8_t*)dst_fb + y * fb_pitch); montauk::memcpy(dst, src, row_bytes); } + + if (hw_fb2) { + // Wait for the latch (vsync): the buffer we just left becomes + // safe to draw into only once the new one is actually scanning. + if (montauk::fb_flip(hw_next, 1) == hw_next) { + hw_next ^= 1; + } + } } }; diff --git a/programs/include/libc/montauk.h b/programs/include/libc/montauk.h index 29a51fd..ace118d 100644 --- a/programs/include/libc/montauk.h +++ b/programs/include/libc/montauk.h @@ -173,6 +173,7 @@ extern "C" { #define MTK_SYS_SDR_SETPARAM 147 #define MTK_SYS_SDR_GETPARAM 148 #define MTK_SYS_POWERINFO 149 +#define MTK_SYS_FBFLIP 150 /* @SYSCALLS-END */ #define MTK_SOCK_TCP 1 diff --git a/programs/include/montauk/syscall.h b/programs/include/montauk/syscall.h index b12b610..dac2896 100644 --- a/programs/include/montauk/syscall.h +++ b/programs/include/montauk/syscall.h @@ -327,6 +327,15 @@ namespace montauk { inline void fb_info(montauk::abi::FbInfo* info) { syscall1(montauk::abi::SYS_FBINFO, (uint64_t)info); } inline void* fb_map() { return (void*)syscall0(montauk::abi::SYS_FBMAP); } + // Page flip between the two scanout buffers. fb_map() maps buffer 1 + // directly after buffer 0 (at +page_align(height*pitch)). index selects + // the buffer to show; the hardware latches it at vblank (tear-free). + // flags bit0 = block until the flip has been latched (vsync). + // fb_flip(-1, 0) returns 1 when page flipping is available, 0 when not. + inline int64_t fb_flip(int64_t index, uint64_t flags) { + return syscall2(montauk::abi::SYS_FBFLIP, (uint64_t)index, flags); + } + // Arguments inline int getargs(char* buf, uint64_t maxLen) { return (int)syscall2(montauk::abi::SYS_GETARGS, (uint64_t)buf, maxLen); diff --git a/programs/libs/libprintersapplet/obj/src/libprintersapplet.o b/programs/libs/libprintersapplet/obj/src/libprintersapplet.o index 00fa983..234bbee 100644 Binary files a/programs/libs/libprintersapplet/obj/src/libprintersapplet.o and b/programs/libs/libprintersapplet/obj/src/libprintersapplet.o differ diff --git a/programs/libs/libtimezonesapplet/obj/src/libtimezonesapplet.o b/programs/libs/libtimezonesapplet/obj/src/libtimezonesapplet.o index 06995a6..89c9dbb 100644 Binary files a/programs/libs/libtimezonesapplet/obj/src/libtimezonesapplet.o and b/programs/libs/libtimezonesapplet/obj/src/libtimezonesapplet.o differ