From 4627ac92fdbceb95145711bb08c5ae29fb228b3c Mon Sep 17 00:00:00 2001 From: Daniel Hammer Date: Thu, 30 Jul 2026 19:28:09 +0100 Subject: [PATCH] fix: correct Intel GGTT setup and DP AUX EDID --- kernel/src/Api/BuildNo.hpp | 2 +- kernel/src/Drivers/Graphics/IntelGPU.cpp | 392 ++++++++++++++++------- kernel/src/Drivers/Graphics/IntelGPU.hpp | 15 + 3 files changed, 288 insertions(+), 121 deletions(-) diff --git a/kernel/src/Api/BuildNo.hpp b/kernel/src/Api/BuildNo.hpp index 2a58c51..7ede517 100644 --- a/kernel/src/Api/BuildNo.hpp +++ b/kernel/src/Api/BuildNo.hpp @@ -12,4 +12,4 @@ #pragma once -#define MONTAUK_BUILD_NUMBER 27 +#define MONTAUK_BUILD_NUMBER 30 diff --git a/kernel/src/Drivers/Graphics/IntelGPU.cpp b/kernel/src/Drivers/Graphics/IntelGPU.cpp index e4a0077..4a00fdc 100644 --- a/kernel/src/Drivers/Graphics/IntelGPU.cpp +++ b/kernel/src/Drivers/Graphics/IntelGPU.cpp @@ -76,6 +76,7 @@ namespace Drivers::Graphics::IntelGPU { static uint32_t g_transTimingBase = TRANS_TIMING_A; static uint32_t g_transConfReg = TRANS_CONF_A; static uint32_t g_transDdiReg = TRANS_DDI_A; + static uint32_t g_auxControlReg = 0; static uint8_t g_connectorType = montauk::abi::DISPLAY_CONNECTOR_UNKNOWN; static bool g_connectorConnected = false; static bool g_edidValid = false; @@ -83,6 +84,12 @@ namespace Drivers::Graphics::IntelGPU { static uint32_t g_pwmCtlReg = 0; static char g_connectorName[32] = "Unknown"; static char g_monitorName[64] = "Unknown display"; + static uint8_t g_auxEdid[256] = {}; + static uint64_t g_auxEdidSize = 0; + static uint32_t g_auxLastStatus = 0; + static uint8_t g_auxLastReply = 0xFF; + + static uint64_t ReadGgttPte(uint64_t index); // ========================================================================= // Register access helpers @@ -426,6 +433,192 @@ namespace Drivers::Graphics::IntelGPU { return sum == 0; } + static uint32_t PackAuxBytes(const uint8_t* bytes, int count) { + uint32_t value = 0; + if (count > 4) count = 4; + for (int i = 0; i < count; i++) + value |= (uint32_t)bytes[i] << ((3 - i) * 8); + return value; + } + + static void UnpackAuxBytes(uint32_t value, uint8_t* bytes, int count) { + if (count > 4) count = 4; + for (int i = 0; i < count; i++) + bytes[i] = (uint8_t)(value >> ((3 - i) * 8)); + } + + // Submit one native or I2C-over-AUX packet. The active firmware display + // has already powered and trained this channel; Montauk only uses it for + // read-only monitor discovery. + static int AuxTransfer(uint8_t request, uint32_t address, + const uint8_t* writeData, int writeSize, + uint8_t* readData, int readCapacity, + uint8_t* reply) { + if (g_auxControlReg == 0 || writeSize < 0 || writeSize > 16 + || readCapacity < 0 || readCapacity > 16) + return -1; + g_auxLastReply = 0xFF; + + uint8_t packet[20] = {}; + packet[0] = (uint8_t)((request << 4) | ((address >> 16) & 0x0F)); + packet[1] = (uint8_t)(address >> 8); + packet[2] = (uint8_t)address; + bool read = (request & 1) != 0; + int payloadSize = read ? readCapacity : writeSize; + int sendBytes = payloadSize ? 4 : 3; + if (payloadSize) packet[3] = (uint8_t)(payloadSize - 1); + if (!read && writeSize) { + memcpy(packet + 4, writeData, (uint64_t)writeSize); + sendBytes += writeSize; + } + + for (int attempt = 0; attempt < 5; attempt++) { + int busyWait = 100000; + while (((g_auxLastStatus = ReadReg(g_auxControlReg)) & DP_AUX_SEND_BUSY) + && --busyWait) + asm volatile("pause"); + if (!busyWait) return -1; + + for (int i = 0; i < sendBytes; i += 4) { + WriteReg(g_auxControlReg + DP_AUX_DATA_DELTA + (uint32_t)(i / 4) * 4, + PackAuxBytes(packet + i, sendBytes - i)); + } + + // Skylake and later derive the AUX clock automatically. The pulse + // lengths below are within the documented hardware ranges. + uint32_t control = DP_AUX_SEND_BUSY | DP_AUX_DONE | DP_AUX_INTERRUPT + | DP_AUX_TIMEOUT_ERROR | DP_AUX_TIMEOUT_MAX + | DP_AUX_RECEIVE_ERROR + | ((uint32_t)sendBytes << DP_AUX_MESSAGE_SHIFT) + | (17u << 5) | 31u; + WriteReg(g_auxControlReg, control); + + int completeWait = 1000000; + uint32_t status; + do { + status = ReadReg(g_auxControlReg); + g_auxLastStatus = status; + if (!(status & DP_AUX_SEND_BUSY)) break; + asm volatile("pause"); + } while (--completeWait); + + if (!completeWait) return -1; + WriteReg(g_auxControlReg, status | DP_AUX_DONE + | DP_AUX_TIMEOUT_ERROR | DP_AUX_RECEIVE_ERROR); + if (!(status & DP_AUX_DONE) + || (status & (DP_AUX_TIMEOUT_ERROR | DP_AUX_RECEIVE_ERROR))) + continue; + + int received = (int)((status >> DP_AUX_MESSAGE_SHIFT) & 0x1F); + if (received <= 0 || received > 20) continue; + + uint8_t response[20] = {}; + for (int i = 0; i < received; i += 4) { + uint32_t value = ReadReg(g_auxControlReg + DP_AUX_DATA_DELTA + + (uint32_t)(i / 4) * 4); + UnpackAuxBytes(value, response + i, received - i); + } + if (reply) *reply = response[0] >> 4; + g_auxLastReply = response[0] >> 4; + if (!read) return writeSize; + + int bytes = received - 1; + if (bytes > readCapacity) bytes = readCapacity; + if (bytes > 0) memcpy(readData, response + 1, (uint64_t)bytes); + return bytes; + } + return -1; + } + + static bool AuxI2cTransfer(uint8_t request, const uint8_t* writeData, + int writeSize, uint8_t* readData, int readSize) { + for (int retry = 0; retry < 16; retry++) { + uint8_t reply = 0xFF; + int result = AuxTransfer(request, 0x50, writeData, writeSize, + readData, readSize, &reply); + uint8_t nativeReply = reply & 0x3; + uint8_t i2cReply = reply & 0xC; + if (result == (readSize ? readSize : writeSize) + && nativeReply == 0 && i2cReply == 0) + return true; + if (nativeReply != 2 && i2cReply != 8 && result >= 0) return false; + for (int delay = 0; delay < 40000; delay++) + asm volatile("pause"); + } + return false; + } + + static bool ReadDpEdidBlock(uint8_t offset, uint8_t* block) { + // Keep the I2C transaction open while setting the EEPROM offset and + // reading the block in the AUX controller's 16-byte payload chunks. + if (!AuxI2cTransfer(0x4, nullptr, 0, nullptr, 0)) return false; + if (!AuxI2cTransfer(0x4, &offset, 1, nullptr, 0)) return false; + if (!AuxI2cTransfer(0x5, nullptr, 0, nullptr, 0)) return false; + for (int pos = 0; pos < 128; pos += 16) { + uint8_t request = (pos + 16 < 128) ? 0x5 : 0x1; + if (!AuxI2cTransfer(request, nullptr, 0, block + pos, 16)) + return false; + } + return true; + } + + static bool ProbeAuxChannel(uint32_t controlReg, uint8_t* revision) { + g_auxControlReg = controlReg; + uint8_t reply = 0xFF; + uint8_t dpcd = 0; + int result = AuxTransfer(0x9, 0, nullptr, 0, &dpcd, 1, &reply); + if (result != 1 || (reply & 0x3) != 0 || dpcd < 0x10 || dpcd > 0x30) + return false; + if (revision) *revision = dpcd; + return true; + } + + static bool ReadDpEdid() { + g_auxEdidSize = 0; + if (g_connectorType != montauk::abi::DISPLAY_CONNECTOR_DP + || g_auxControlReg == 0) + return false; + + uint32_t preferred = g_auxControlReg; + uint32_t candidates[] = { + preferred, DP_AUX_A_CTL, DP_AUX_B_CTL, DP_AUX_C_CTL, DP_AUX_D_CTL, + }; + for (uint32_t i = 0; i < sizeof(candidates) / sizeof(candidates[0]); i++) { + bool duplicate = false; + for (uint32_t j = 0; j < i; j++) { + if (candidates[j] == candidates[i]) duplicate = true; + } + if (duplicate) continue; + + uint8_t revision = 0; + if (!ProbeAuxChannel(candidates[i], &revision)) { + KernelLogStream(DEBUG, "IntelGPU") << "DP AUX probe failed at " + << base::hex << candidates[i] << " (status=" << g_auxLastStatus + << ", reply=" << (uint64_t)g_auxLastReply << ")"; + continue; + } + + KernelLogStream(INFO, "IntelGPU") << "DP AUX channel at " + << base::hex << candidates[i] << ", DPCD revision " + << (uint64_t)revision; + if (!ReadDpEdidBlock(0, g_auxEdid) || !EdidBlockValid(g_auxEdid)) { + KernelLogStream(WARNING, "IntelGPU") << "EDID read failed on DP AUX " + << base::hex << candidates[i] << " (status=" << g_auxLastStatus + << ", reply=" << (uint64_t)g_auxLastReply << ")"; + continue; + } + + g_auxEdidSize = 128; + if (g_auxEdid[126] != 0 + && ReadDpEdidBlock(128, g_auxEdid + 128) + && EdidBlockValid(g_auxEdid + 128)) + g_auxEdidSize = 256; + return true; + } + g_auxControlReg = preferred; + return false; + } + static bool ParseDetailedTiming(const uint8_t* dtd, DisplayMode& mode) { uint32_t clock10KHz = (uint32_t)dtd[0] | ((uint32_t)dtd[1] << 8); if (clock10KHz == 0) return false; @@ -460,7 +653,17 @@ namespace Drivers::Graphics::IntelGPU { static void ParseEdid() { uint64_t edidSize = 0; const uint8_t* edid = ::Graphics::Framebuffer::GetEdid(&edidSize); - if (!edid || edidSize < 128) return; + if (!edid || edidSize < 128) { + if (!ReadDpEdid()) { + KernelLogStream(INFO, "IntelGPU") << "EDID unavailable from boot data" + << (g_auxControlReg ? " and DP AUX" : ""); + return; + } + edid = g_auxEdid; + edidSize = g_auxEdidSize; + KernelLogStream(OK, "IntelGPU") << "Read " << base::dec << edidSize + << " bytes of EDID over DP AUX"; + } static constexpr uint8_t header[8] = {0x00, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0x00}; if (memcmp(edid, header, sizeof(header)) != 0 || !EdidBlockValid(edid)) { @@ -560,6 +763,7 @@ namespace Drivers::Graphics::IntelGPU { g_transDdiReg = trans.ddi; g_connectorConnected = (ReadReg(trans.conf) & PIPECONF_ENABLE) != 0; uint32_t ddi = ReadReg(trans.ddi); + g_auxControlReg = 0; if (g_gpuGen <= 7) { if (ReadReg(LVDS) & LVDS_PORT_ENABLE) { @@ -595,6 +799,11 @@ namespace Drivers::Graphics::IntelGPU { } else { g_connectorType = montauk::abi::DISPLAY_CONNECTOR_DP; CopyText(g_connectorName, sizeof(g_connectorName), "DP-"); + static constexpr uint32_t auxControls[] = { + DP_AUX_A_CTL, DP_AUX_B_CTL, DP_AUX_C_CTL, DP_AUX_D_CTL, + }; + if (port < sizeof(auxControls) / sizeof(auxControls[0])) + g_auxControlReg = auxControls[port]; } uint64_t len = 0; while (len + 1 < sizeof(g_connectorName) && g_connectorName[len]) len++; @@ -720,22 +929,17 @@ namespace Drivers::Graphics::IntelGPU { uint8_t dev = g_gpuInfo.pciDevice; uint8_t func = g_gpuInfo.pciFunction; - // Read GMCH_CTL to determine GTT size + // Read GGC/GMCH_CTL to determine the GGTT modification window. uint16_t gmchCtl = Pci::LegacyRead16(bus, dev, func, (uint8_t)PCI_REG_GMCH_CTL); - uint8_t gttSizeBits = (gmchCtl >> 8) & 0x3; - uint64_t gttSizeBytes = 0; + uint64_t gttOffset = 0; if (g_gpuGen >= 8) { - // Gen 8+ has a different encoding for GTT size - switch (gttSizeBits) { - case 0: gttSizeBytes = 0; break; // No GTT - case 1: gttSizeBytes = 2 * 1024 * 1024; break; // 2MB - case 2: gttSizeBytes = 4 * 1024 * 1024; break; // 4MB - case 3: gttSizeBytes = 8 * 1024 * 1024; break; // 8MB - default: gttSizeBytes = 2 * 1024 * 1024; break; - } + uint8_t ggms = (gmchCtl >> 6) & 0x3; + if (ggms != 0) gttSizeBytes = (1ull << ggms) * 1024 * 1024; + gttOffset = 8 * 1024 * 1024; } else { + uint8_t gttSizeBits = (gmchCtl >> 8) & 0x3; // Gen 5-7 switch (gttSizeBits) { case 0: gttSizeBytes = 0; break; // No GTT @@ -744,12 +948,13 @@ namespace Drivers::Graphics::IntelGPU { case 3: gttSizeBytes = 2 * 1024 * 1024; break; // Depends on gen, default 2MB default: gttSizeBytes = 1024 * 1024; break; } + gttOffset = 2 * 1024 * 1024; } if (gttSizeBytes == 0) { - // If hardware reports no GTT, assume 1MB as a safe fallback - gttSizeBytes = 1024 * 1024; - KernelLogStream(WARNING, "IntelGPU") << "GMCH_CTL reports no GTT, assuming 1MB"; + KernelLogStream(ERROR, "IntelGPU") << "GGC reports no GGTT (GGC=" + << base::hex << (uint64_t)gmchCtl << ")"; + return false; } g_gpuInfo.gttSize = gttSizeBytes; @@ -757,9 +962,7 @@ namespace Drivers::Graphics::IntelGPU { KernelLogStream(INFO, "IntelGPU") << "GMCH_CTL: " << base::hex << (uint64_t)gmchCtl << ", GTT size: " << base::dec << (gttSizeBytes / 1024) << " KB"; - // The GTT entries reside at BAR0 + 2MB (offset 0x200000) - // This is correct for most Intel generations - uint64_t gttPhys = g_gpuInfo.mmioPhys + 0x200000; + uint64_t gttPhys = g_gpuInfo.mmioPhys + gttOffset; // Map the GTT region (it may overlap with already-mapped MMIO, but we map // additional pages beyond the initial 2MB MMIO mapping) @@ -770,6 +973,7 @@ namespace Drivers::Graphics::IntelGPU { } g_gttBase = (volatile void*)Memory::HHDM(gttPhys); + g_ggtt = g_gttBase; // Calculate number of GTT entries if (g_gpuGen >= 8) { @@ -779,6 +983,7 @@ namespace Drivers::Graphics::IntelGPU { // 32-bit PTEs g_gttEntryCount = gttSizeBytes / sizeof(uint32_t); } + g_ggttEntries = g_gttEntryCount; KernelLogStream(INFO, "IntelGPU") << "GTT at physical " << base::hex << gttPhys << ", " << base::dec << g_gttEntryCount << " entries" @@ -804,10 +1009,10 @@ namespace Drivers::Graphics::IntelGPU { // ========================================================================= static bool SetupFramebuffer() { - // Map the firmware framebuffer's contiguous physical pages through our GTT. - // This keeps the same physical memory that the firmware set up (contiguous - // pages, already HHDM-mapped), so both kernel and userspace access continue - // to work via the original virtual/physical addresses. No copy is needed. + // The boot framebuffer can be the CPU-visible graphics aperture rather + // than the scanout memory itself. On Gen 8+ retain the firmware's GGTT + // mapping and its surface offset; remapping the aperture address as a + // backing page would create an invalid translation. uint32_t* fwFb = ::Graphics::Framebuffer::GetBase(); if (fwFb == nullptr) { KernelLogStream(ERROR, "IntelGPU") << "No firmware framebuffer available"; @@ -823,33 +1028,36 @@ namespace Drivers::Graphics::IntelGPU { return false; } - KernelLogStream(INFO, "IntelGPU") << "Mapping " << base::dec << pageCount - << " firmware FB pages through GTT (phys base " << base::hex << fwFbPhys << ")"; - - // Program GTT entries to point to the firmware FB's contiguous physical pages if (g_gpuGen >= 8) { - volatile uint64_t* gtt64 = (volatile uint64_t*)g_gttBase; - for (uint64_t i = 0; i < pageCount; i++) { - gtt64[i] = MakeGttPte64(fwFbPhys + i * 0x1000); + g_fbGttOffset = ReadReg(DSPASURF) & ~0xFFFull; + uint64_t first = g_fbGttOffset >> 12; + uint64_t last = (g_fbGttOffset + g_fbSize - 1) >> 12; + if (last >= g_gttEntryCount || !(ReadGgttPte(first) & 1) + || !(ReadGgttPte(last) & 1)) { + KernelLogStream(ERROR, "IntelGPU") + << "Firmware scanout is not backed by valid GGTT entries"; + return false; } - // Flush GTT writes - (void)gtt64[pageCount - 1]; + KernelLogStream(OK, "IntelGPU") << "Preserving firmware GGTT scanout at +" + << base::hex << g_fbGttOffset << " (CPU aperture " << fwFbPhys << ")"; } else { + KernelLogStream(INFO, "IntelGPU") << "Mapping " << base::dec << pageCount + << " firmware FB pages through GTT (phys base " + << base::hex << fwFbPhys << ")"; volatile uint32_t* gtt32 = (volatile uint32_t*)g_gttBase; for (uint64_t i = 0; i < pageCount; i++) { gtt32[i] = MakeGttPte32(fwFbPhys + i * 0x1000); } // Flush GTT writes (void)gtt32[pageCount - 1]; + g_fbGttOffset = 0; } // Keep using the same framebuffer memory g_fbBase = fwFb; g_fbPhysBase = fwFbPhys; - g_fbGttOffset = 0; // Starting at GTT entry 0 => offset 0 - - KernelLogStream(OK, "IntelGPU") << "Framebuffer mapped through GTT: " << base::dec << pageCount - << " pages, phys=" << base::hex << fwFbPhys; + KernelLogStream(OK, "IntelGPU") << "Framebuffer ready: " << base::dec << pageCount + << " pages, CPU phys=" << base::hex << fwFbPhys; return true; } @@ -877,8 +1085,7 @@ namespace Drivers::Graphics::IntelGPU { // in the hardware's native format (bytes on Gen <9, 64-byte units on Gen 9+). // Writing our byte-converted g_fbPitch would corrupt it on Gen 9+. - // Write the GTT base offset to DSPASURF - this triggers the plane update - // Since we mapped at GTT entry 0, the offset is 0 + // Reassert the firmware surface offset to flush any plane update. uint32_t surfAddr = (uint32_t)g_fbGttOffset; WriteReg(DSPASURF, surfAddr); @@ -898,14 +1105,8 @@ namespace Drivers::Graphics::IntelGPU { // The GTT window lives in the upper half of GTTMMADR (BAR0): // Gen 6-7: 4 MB BAR, GTT at +2 MB, 32-bit entries // Gen 8+: 16 MB BAR, GGTT at +8 MB, 64-bit entries - // The legacy init path above writes at +2 MB for all generations; on - // Gen 8+ those writes land in register space, and scanout keeps working - // only because the firmware's real GGTT entries (stored in stolen RAM, - // which survives S3 in self-refresh) are never actually touched. Page - // flipping needs entries the display engine really reads, so it uses the - // correct window and refuses to run unless the firmware's own scanout - // PTEs are visible there (valid bits set). On any failure page flipping - // stays off and the driver behaves exactly as before. + // Initialization maps this window once and preserves the firmware's active + // scanout PTEs. Page flipping only proceeds after those entries validate. static void MicroDelay(int us) { // Simple busy-wait; us is approximate @@ -928,35 +1129,9 @@ namespace Drivers::Graphics::IntelGPU { } static bool MapCorrectGgtt() { - if (g_gpuGen < 8) { - // The legacy mapping at BAR0 + 2 MB is already the real GTT - g_ggtt = g_gttBase; - g_ggttEntries = g_gttEntryCount; - return g_ggtt != nullptr; - } - - // Gen 8+: GGC.GGMS moved to bits 7:6 (0 = none, 1/2/3 = 2/4/8 MB) - uint16_t gmchCtl = Pci::LegacyRead16(g_gpuInfo.pciBus, g_gpuInfo.pciDevice, - g_gpuInfo.pciFunction, (uint8_t)PCI_REG_GMCH_CTL); - uint8_t ggms = (gmchCtl >> 6) & 0x3; - if (ggms == 0) { - KernelLogStream(WARNING, "IntelGPU") << "GGC reports no GGTT (GGC=" - << base::hex << (uint64_t)gmchCtl << "), page flip unavailable"; - return false; - } - uint64_t ggttBytes = (1ull << ggms) * 1024 * 1024; - uint64_t ggttPhys = g_gpuInfo.mmioPhys + 8 * 1024 * 1024; - - for (uint64_t off = 0; off < ggttBytes; off += 0x1000) { - Memory::VMM::g_paging->MapMMIO(ggttPhys + off, Memory::HHDM(ggttPhys + off)); - } - - g_ggtt = (volatile void*)Memory::HHDM(ggttPhys); - g_ggttEntries = ggttBytes / sizeof(uint64_t); - - KernelLogStream(INFO, "IntelGPU") << "GGTT window at BAR0+8MB (phys " - << base::hex << ggttPhys << "), " << base::dec << (ggttBytes / 1024) - << " KB, " << g_ggttEntries << " entries"; + if (!g_gttBase || g_gttEntryCount == 0) return false; + g_ggtt = g_gttBase; + g_ggttEntries = g_gttEntryCount; return true; } @@ -1017,45 +1192,25 @@ namespace Drivers::Graphics::IntelGPU { static bool FindAndMapGgttRange() { uint64_t pages = (g_fbSize + 0xFFF) >> 12; constexpr uint64_t guard = 16; - uint64_t need = pages + 2 * guard; - - // Firmware does not leave unused GGTT entries invalid: it points the - // ENTIRE table at a scratch page, so free entries are valid but all - // hold one identical PTE value (HW-confirmed on Raptor Lake, where a - // scan for invalid entries found none in 1M entries). Sample the tail - // of the table to learn the scratch value; a non-uniform tail means - // the layout is not what we expect, so bail. - uint64_t scratchPte = ReadGgttPte(g_ggttEntries - 1); - for (uint64_t probe = 2; probe <= 32; probe++) { - if (ReadGgttPte(g_ggttEntries - probe) != scratchPte) { - KernelLogStream(WARNING, "IntelGPU") << "GGTT tail not uniform (entry -" - << base::dec << probe << " != " << base::hex << scratchPte - << "), page flip unavailable"; - return false; - } - } - - // Search the upper half for a run of scratch-backed (or invalid) - // entries. Never overlap the active scanout range. uint64_t avoidFirst = g_fbGttOffsetA >> 12; uint64_t avoidLast = (g_fbGttOffsetA + g_fbSize - 1) >> 12; - uint64_t runStart = 0, run = 0; - bool found = false; - for (uint64_t idx = g_ggttEntries / 2; idx < g_ggttEntries; idx++) { - if (idx >= avoidFirst && idx <= avoidLast) { run = 0; continue; } - uint64_t pte = ReadGgttPte(idx); - if ((pte & 1) && pte != scratchPte) { run = 0; continue; } - if (run == 0) runStart = idx; - if (++run >= need) { found = true; break; } - } - if (!found) { - KernelLogStream(WARNING, "IntelGPU") << "No free GGTT run of " << base::dec - << need << " entries in upper half, page flip unavailable"; + // Montauk owns the GPU after firmware handoff. Reserve a deterministic + // high-aperture range for its second scanout buffer instead of trying + // to infer firmware ownership from scratch-PTE patterns. Keep guards + // around the allocation and never overlap the live firmware surface. + uint64_t base = g_ggttEntries / 2 + guard; + if (base <= avoidLast + guard) base = avoidLast + guard + 1; + base = (base + guard - 1) & ~(guard - 1); + if (base < guard || base + pages + guard > g_ggttEntries + || !(base + pages - 1 < avoidFirst || base > avoidLast)) { + KernelLogStream(WARNING, "IntelGPU") << "No safe owned GGTT range for " + << base::dec << pages << " back-buffer pages"; return false; } - uint64_t base = runStart + guard; + uint64_t replacedFirst = ReadGgttPte(base); + uint64_t replacedLast = ReadGgttPte(base + pages - 1); for (uint64_t i = 0; i < pages; i++) { WriteGgttPte(base + i, g_buf1Phys + i * 0x1000); } @@ -1073,9 +1228,10 @@ namespace Drivers::Graphics::IntelGPU { } g_buf1GttOffset = base << 12; - KernelLogStream(OK, "IntelGPU") << "Buffer 1 mapped at GGTT+" << base::hex + KernelLogStream(OK, "IntelGPU") << "Claimed buffer 1 at GGTT+" << base::hex << g_buf1GttOffset << " (" << base::dec << pages << " pages, phys " - << base::hex << g_buf1Phys << ")"; + << base::hex << g_buf1Phys << ", replaced PTEs " + << replacedFirst << ".." << replacedLast << ")"; return true; } @@ -1679,16 +1835,11 @@ namespace Drivers::Graphics::IntelGPU { // 2. Disable VGA plane (firmware may have re-enabled it during POST) DisableVga(); - // 3. Reprogram GTT entries (hardware lost all GTT state during S3) + // 3. Restore legacy framebuffer mappings. Gen 8+ firmware scanout uses + // an aperture mapping whose backing physical pages are not described + // by the boot framebuffer address, so preserve the restored entries. uint64_t pageCount = (g_fbSize + 0xFFF) / 0x1000; - if (g_gpuGen >= 8) { - volatile uint64_t* gtt64 = (volatile uint64_t*)g_gttBase; - for (uint64_t i = 0; i < pageCount; i++) { - gtt64[i] = MakeGttPte64(g_fbPhysBase + i * 0x1000); - } - // Flush GTT writes by reading back the last entry - (void)gtt64[pageCount - 1]; - } else { + if (g_gpuGen < 8) { volatile uint32_t* gtt32 = (volatile uint32_t*)g_gttBase; for (uint64_t i = 0; i < pageCount; i++) { gtt32[i] = MakeGttPte32(g_fbPhysBase + i * 0x1000); @@ -1697,8 +1848,9 @@ namespace Drivers::Graphics::IntelGPU { (void)gtt32[pageCount - 1]; } - KernelLogStream(DEBUG, "IntelGPU") << "GTT reprogrammed: " << base::dec << pageCount - << " pages (" << (g_gpuGen >= 8 ? "64-bit" : "32-bit") << " PTEs)"; + KernelLogStream(DEBUG, "IntelGPU") << (g_gpuGen >= 8 + ? "Firmware GGTT scanout preserved" + : "Legacy GTT framebuffer mappings restored"); // 4. Re-enable the display pipe. After S3, the pipe may be off even // though the firmware lit the backlight. Wait for it to become diff --git a/kernel/src/Drivers/Graphics/IntelGPU.hpp b/kernel/src/Drivers/Graphics/IntelGPU.hpp index 163e9d5..5d9c7d8 100644 --- a/kernel/src/Drivers/Graphics/IntelGPU.hpp +++ b/kernel/src/Drivers/Graphics/IntelGPU.hpp @@ -284,6 +284,21 @@ namespace Drivers::Graphics::IntelGPU { static constexpr uint32_t TRANS_DDI_PORT_MASK = (0xFu << 27); static constexpr uint32_t TRANS_DDI_MODE_MASK = (0x7u << 24); + // DisplayPort AUX channels. Each port has one control register followed + // by five big-endian data registers. + static constexpr uint32_t DP_AUX_A_CTL = 0x64010; + static constexpr uint32_t DP_AUX_B_CTL = 0x64110; + static constexpr uint32_t DP_AUX_C_CTL = 0x64210; + static constexpr uint32_t DP_AUX_D_CTL = 0x64310; + static constexpr uint32_t DP_AUX_DATA_DELTA = 4; + static constexpr uint32_t DP_AUX_SEND_BUSY = (1u << 31); + static constexpr uint32_t DP_AUX_DONE = (1u << 30); + static constexpr uint32_t DP_AUX_INTERRUPT = (1u << 29); + static constexpr uint32_t DP_AUX_TIMEOUT_ERROR = (1u << 28); + static constexpr uint32_t DP_AUX_TIMEOUT_MAX = (3u << 26); + static constexpr uint32_t DP_AUX_RECEIVE_ERROR = (1u << 25); + static constexpr uint32_t DP_AUX_MESSAGE_SHIFT = 20; + // PCH/CPU panel PWM. Firmware owns the frequency; the driver only changes // the duty field and preserves enable/polarity/reserved bits. static constexpr uint32_t BLC_PWM_CPU_CTL2 = 0x48250;