Qt
Internal/Contributor docs for the Qt SDK. Note: These are NOT official API docs; those are found at https://doc.qt.io/
Loading...
Searching...
No Matches
qrhimetal.mm
Go to the documentation of this file.
1// Copyright (C) 2023 The Qt Company Ltd.
2// SPDX-License-Identifier: LicenseRef-Qt-Commercial OR LGPL-3.0-only OR GPL-2.0-only OR GPL-3.0-only
3// Qt-Security score:significant reason:default
4
5#include "qrhimetal_p.h"
7#include "qshader_p.h"
8#include <QGuiApplication>
9#include <QWindow>
10#include <QUrl>
11#include <QFile>
12#include <QTemporaryFile>
13#include <QFileInfo>
14#include <qmath.h>
15#include <QOperatingSystemVersion>
16
17#include <QtCore/private/qcore_mac_p.h>
18#include <QtGui/private/qmetallayer_p.h>
19#include <QtGui/qpa/qplatformwindow_p.h>
20
21#ifdef Q_OS_MACOS
22#include <AppKit/AppKit.h>
23#else
24#include <UIKit/UIKit.h>
25#endif
26
27#include <QuartzCore/CATransaction.h>
28
29#include <Metal/Metal.h>
30
31#include <utility> // for std::pair
32
34
35/*
36 Metal backend. Double buffers and throttles to vsync. "Dynamic" buffers are
37 Shared (host visible) and duplicated (to help having 2 frames in flight),
38 "static" and "immutable" are Managed on macOS and Shared on iOS/tvOS.
39 Textures are Private (device local) and a host visible staging buffer is
40 used to upload data to them. Does not rely on strong objects refs from
41 command buffers but does rely on the automatic resource tracking of the
42 command encoders. Assumes that an autorelease pool (ideally per frame) is
43 available on the thread on which QRhi is used.
44*/
45
46#if __has_feature(objc_arc)
47#error ARC not supported
48#endif
49
50// Even though the macOS 13 MTLBinaryArchive problem (QTBUG-106703) seems
51// to be solved in later 13.x releases, we have reports from old Intel hardware
52// and older macOS versions where this causes problems (QTBUG-114338).
53// Thus we no longer do OS version based differentiation, but rather have a
54// single toggle that is currently on, and so QRhi::(set)pipelineCache()
55// does nothing with Metal.
56#define QRHI_METAL_DISABLE_BINARY_ARCHIVE
57
58// We should be able to operate with command buffers that do not automatically
59// retain/release the resources used by them. (since we have logic that mirrors
60// other backends such as the Vulkan one anyway)
61#define QRHI_METAL_COMMAND_BUFFERS_WITH_UNRETAINED_REFERENCES
62
63/*!
64 \class QRhiMetalInitParams
65 \inmodule QtGuiPrivate
66 \inheaderfile rhi/qrhi.h
67 \since 6.6
68 \brief Metal specific initialization parameters.
69
70 \note This is a RHI API with limited compatibility guarantees, see \l QRhi
71 for details.
72
73 A Metal-based QRhi needs no special parameters for initialization.
74
75 \badcode
76 QRhiMetalInitParams params;
77 rhi = QRhi::create(QRhi::Metal, &params);
78 \endcode
79
80 \note Metal API validation cannot be enabled programmatically by the QRhi.
81 Instead, either run the debug build of the application in XCode, by
82 generating a \c{.xcodeproj} file via \c{cmake -G Xcode}, or set the
83 environment variable \c{METAL_DEVICE_WRAPPER_TYPE=1}. The variable needs to
84 be set early on in the environment, preferably before starting the process;
85 attempting to set it at QRhi creation time is not functional in practice.
86 (too late probably)
87
88 \note QRhiSwapChain can only target QWindow instances that have their
89 surface type set to QSurface::MetalSurface.
90
91 \section2 Working with existing Metal devices
92
93 When interoperating with another graphics engine, it may be necessary to
94 get a QRhi instance that uses the same Metal device. This can be achieved
95 by passing a pointer to a QRhiMetalNativeHandles to QRhi::create(). The
96 device must be set to a non-null value then. Optionally, a command queue
97 object can be specified as well.
98
99 The QRhi does not take ownership of any of the external objects.
100 */
101
102/*!
103 \class QRhiMetalNativeHandles
104 \inmodule QtGuiPrivate
105 \inheaderfile rhi/qrhi.h
106 \since 6.6
107 \brief Holds the Metal device used by the QRhi.
108
109 \note This is a RHI API with limited compatibility guarantees, see \l QRhi
110 for details.
111 */
112
113/*!
114 \variable QRhiMetalNativeHandles::dev
115
116 Set to a valid MTLDevice to import an existing device.
117*/
118
119/*!
120 \variable QRhiMetalNativeHandles::cmdQueue
121
122 Set to a valid MTLCommandQueue when importing an existing command queue.
123 When \nullptr, QRhi will create a new command queue.
124*/
125
126/*!
127 \class QRhiMetalCommandBufferNativeHandles
128 \inmodule QtGuiPrivate
129 \inheaderfile rhi/qrhi.h
130 \since 6.6
131 \brief Holds the MTLCommandBuffer and MTLRenderCommandEncoder objects that are backing a QRhiCommandBuffer.
132
133 \note The command buffer object is only guaranteed to be valid while
134 recording a frame, that is, between a \l{QRhi::beginFrame()}{beginFrame()}
135 - \l{QRhi::endFrame()}{endFrame()} or
136 \l{QRhi::beginOffscreenFrame()}{beginOffscreenFrame()} -
137 \l{QRhi::endOffscreenFrame()}{endOffscreenFrame()} pair.
138
139 \note The command encoder is only valid while recording a pass, that is,
140 between \l{QRhiCommandBuffer::beginPass()} -
141 \l{QRhiCommandBuffer::endPass()}.
142
143 \note This is a RHI API with limited compatibility guarantees, see \l QRhi
144 for details.
145 */
146
147/*!
148 \variable QRhiMetalCommandBufferNativeHandles::commandBuffer
149*/
150
151/*!
152 \variable QRhiMetalCommandBufferNativeHandles::encoder
153*/
154
156{
159 std::array<uint, 3> localSize = {};
166
167 void destroy() {
168 nativeResourceBindingMap.clear();
169 [lib release];
170 lib = nil;
171 [func release];
172 func = nil;
173 [argumentEncoder release];
174 argumentEncoder = nil;
176 }
177};
178
180{
181 QRhiMetalData(QRhiMetal *rhi) : q(rhi), ofr(rhi) { }
182
187
190 const QColor &colorClearValue,
191 const QRhiDepthStencilClearValue &depthStencilClearValue,
192 int colorAttCount,
193 QRhiShadingRateMap *shadingRateMap);
194 id<MTLLibrary> createMetalLib(const QShader &shader, QShader::Variant shaderVariant,
195 bool preferArgumentBuffers,
196 QString *error, QByteArray *entryPoint, QShaderKey *activeKey);
197 id<MTLFunction> createMSLShaderFunction(id<MTLLibrary> lib, const QByteArray &entryPoint);
198 bool setupBinaryArchive(NSURL *sourceFileUrl = nil);
199 void addRenderPipelineToBinaryArchive(MTLRenderPipelineDescriptor *rpDesc);
200 void trySeedingRenderPipelineFromBinaryArchive(MTLRenderPipelineDescriptor *rpDesc);
201 void addComputePipelineToBinaryArchive(MTLComputePipelineDescriptor *cpDesc);
202 void trySeedingComputePipelineFromBinaryArchive(MTLComputePipelineDescriptor *cpDesc);
203
257
259 OffscreenFrame(QRhiImplementation *rhi) : cbWrapper(rhi) { }
260 bool active = false;
261 double lastGpuTime = 0;
263 } ofr;
264
275
283
285
288
289 // Indirect Command Buffer (ICB) infrastructure for GPU-driven multi-draw
299 // Written by the encode kernel, consumed by executeCommandsInBuffer.
301 // Holds 0xFFFFFFFF, stands in for the count buffer when there is none.
303 bool icbSetupFailed = false;
304
305 static const int TEXBUF_ALIGN = 256; // probably not accurate
306
307 using ShaderCacheKey = std::pair<QRhiShaderStage, bool>; // stage, argument_buffers
309
320 id<MTLBuffer> allocFromStagingArea(StagingArea *area, quint32 size, quint32 alignment,
321 int frameSlot, quint32 minBlockSize, quint32 *offset);
322 id<MTLBuffer> allocArgumentBuffer(quint32 size, quint32 alignment, int frameSlot, quint32 *offset);
323 id<MTLBuffer> allocBufferStaging(quint32 size, int frameSlot, quint32 *offset);
324 id<MTLBuffer> newOneShotStagingBuffer(quint32 size, int frameSlot);
326
327 // Uploads larger than this always get their own one-shot MTLBuffer and do
328 // not count towards the area's size, so that a one-off large upload cannot
329 // inflate it.
330 static constexpr quint32 LARGE_STAGING_ALLOC = 512 * 1024;
331 static constexpr quint32 STAGING_AREA_MIN = 64 * 1024;
332 static constexpr quint32 STAGING_AREA_MAX = 16 * 1024 * 1024;
333 static constexpr int STAGING_AREA_LOW_DEMAND_FRAMES = 60;
334 static constexpr int STAGING_AREA_HIGH_DEMAND_FRAMES = 3;
335
336 // Counts both swapchain and offscreen frames. Never 0 once a frame started,
337 // so that a default-initialized "used in frame" cannot match.
339};
340
343
345{
346 bool managed = false;
347 bool slotted = false;
348 bool isDeviceLocal = false; // true = MTLResourceStorageModePrivate buf[0], no slotting, non-host visible
355};
356
362
384
389
394
422
445
478
480{
484 bool icbCapable = false;
505 bool enabled = false;
506 bool failed = false;
514 quint32 vsCompOutputBufferSize(quint32 vertexOrIndexCount, quint32 instanceCount) const
515 {
516 // max vertex output components = resourceLimit(MaxVertexOutputs) * 4 = 60
517 return vertexOrIndexCount * instanceCount * sizeof(float) * 60;
518 }
519 quint32 tescCompOutputBufferSize(quint32 patchCount) const
520 {
521 return outControlPointCount * patchCount * sizeof(float) * 60;
522 }
523 quint32 tescCompPatchOutputBufferSize(quint32 patchCount) const
524 {
525 // assume maxTessellationControlPerPatchOutputComponents is 128
526 return patchCount * sizeof(float) * 128;
527 }
528 quint32 patchCountForDrawCall(quint32 vertexOrIndexCount, quint32 instanceCount) const
529 {
530 return ((vertexOrIndexCount + inControlPointCount - 1) / inControlPointCount) * instanceCount;
531 }
536 } tess;
537 void setupVertexInputDescriptor(MTLVertexDescriptor *desc);
538 void setupStageInputDescriptor(MTLStageInputOutputDescriptor *desc);
539
540 // SPIRV-Cross buffer size buffers
542};
543
545{
549
550 // SPIRV-Cross buffer size buffers
552};
553
565
566QRhiMetal::QRhiMetal(QRhiMetalInitParams *params, QRhiMetalNativeHandles *importDevice)
567{
568 Q_UNUSED(params);
569
570 d = new QRhiMetalData(this);
571
572 importedDevice = importDevice != nullptr;
573 if (importedDevice) {
574 if (importDevice->dev) {
575 d->dev = (id<MTLDevice>) importDevice->dev;
576 importedCmdQueue = importDevice->cmdQueue != nullptr;
577 if (importedCmdQueue)
578 d->cmdQueue = (id<MTLCommandQueue>) importDevice->cmdQueue;
579 } else {
580 qWarning("No MTLDevice given, cannot import");
581 importedDevice = false;
582 }
583 }
584}
585
587{
588 delete d;
589}
590
591template <class Int>
592inline Int aligned(Int v, Int byteAlign)
593{
594 return (v + byteAlign - 1) & ~(byteAlign - 1);
595}
596
597bool QRhiMetal::probe(QRhiMetalInitParams *params)
598{
599 QMacAutoReleasePool pool;
600
601 Q_UNUSED(params);
602 id<MTLDevice> dev = MTLCreateSystemDefaultDevice();
603 if (dev) {
604 [dev release];
605 return true;
606 }
607 return false;
608}
609
611{
613 // Do not let the command buffer mess with the refcount of objects. We do
614 // have a proper render loop and will manage lifetimes similarly to other
615 // backends (Vulkan).
616 return [cmdQueue commandBufferWithUnretainedReferences];
617#else
618 return [cmdQueue commandBuffer];
619#endif
620}
621
622bool QRhiMetalData::setupBinaryArchive(NSURL *sourceFileUrl)
623{
625 return false;
626#endif
627
628 [binArch release];
629 MTLBinaryArchiveDescriptor *binArchDesc = [MTLBinaryArchiveDescriptor new];
630 binArchDesc.url = sourceFileUrl;
631 NSError *err = nil;
632 binArch = [dev newBinaryArchiveWithDescriptor: binArchDesc error: &err];
633 [binArchDesc release];
634 if (!binArch) {
635 const QString msg = QString::fromNSString(err.localizedDescription);
636 qWarning("newBinaryArchiveWithDescriptor failed: %s", qPrintable(msg));
637 return false;
638 }
639 return true;
640}
641
642bool QRhiMetal::create(QRhi::Flags flags)
643{
644 rhiFlags = flags;
645
646 if (importedDevice)
647 [d->dev retain];
648 else
649 d->dev = MTLCreateSystemDefaultDevice();
650
651 if (!d->dev) {
652 qWarning("No MTLDevice");
653 return false;
654 }
655
656 const QString deviceName = QString::fromNSString([d->dev name]);
657 qCDebug(QRHI_LOG_INFO, "Metal device: %s", qPrintable(deviceName));
658 driverInfoStruct.deviceName = deviceName.toUtf8();
659
660 // deviceId and vendorId stay unset for now. Note that registryID is not
661 // suitable as deviceId because it does not seem stable on macOS and can
662 // apparently change when the system is rebooted.
663
664#ifdef Q_OS_MACOS
665 const MTLDeviceLocation deviceLocation = [d->dev location];
666 switch (deviceLocation) {
667 case MTLDeviceLocationBuiltIn:
668 driverInfoStruct.deviceType = QRhiDriverInfo::IntegratedDevice;
669 break;
670 case MTLDeviceLocationSlot:
671 driverInfoStruct.deviceType = QRhiDriverInfo::DiscreteDevice;
672 break;
673 case MTLDeviceLocationExternal:
674 driverInfoStruct.deviceType = QRhiDriverInfo::ExternalDevice;
675 break;
676 default:
677 break;
678 }
679#else
680 driverInfoStruct.deviceType = QRhiDriverInfo::IntegratedDevice;
681#endif
682
683 const QOperatingSystemVersion ver = QOperatingSystemVersion::current();
684 osMajor = ver.majorVersion();
685 osMinor = ver.minorVersion();
686
687 if (importedCmdQueue)
688 [d->cmdQueue retain];
689 else
690 d->cmdQueue = [d->dev newCommandQueue];
691
692 d->captureMgr = [MTLCaptureManager sharedCaptureManager];
693 // Have a custom capture scope as well which then shows up in XCode as
694 // an option when capturing, and becomes especially useful when having
695 // multiple windows with multiple QRhis.
696 d->captureScope = [d->captureMgr newCaptureScopeWithCommandQueue: d->cmdQueue];
697 const QString label = QString::asprintf("Qt capture scope for QRhi %p", this);
698 d->captureScope.label = label.toNSString();
699
700#if defined(Q_OS_MACOS) || defined(Q_OS_VISIONOS)
701 caps.maxTextureSize = 16384;
702 caps.baseVertexAndInstance = true;
703 caps.isAppleGPU = [d->dev supportsFamily:MTLGPUFamilyApple7];
704 caps.maxThreadGroupSize = 1024;
705 caps.multiView = true;
706#elif defined(Q_OS_TVOS)
707 if ([d->dev supportsFamily:MTLGPUFamilyApple3])
708 caps.maxTextureSize = 16384;
709 else
710 caps.maxTextureSize = 8192;
711 caps.baseVertexAndInstance = false;
712 caps.isAppleGPU = true;
713#elif defined(Q_OS_IOS)
714 if ([d->dev supportsFamily:MTLGPUFamilyApple3]) {
715 caps.maxTextureSize = 16384;
716 caps.baseVertexAndInstance = true;
717 } else if ([d->dev supportsFamily:MTLGPUFamilyApple2]) {
718 caps.maxTextureSize = 8192;
719 caps.baseVertexAndInstance = false;
720 } else {
721 caps.maxTextureSize = 4096;
722 caps.baseVertexAndInstance = false;
723 }
724 caps.isAppleGPU = true;
725 if ([d->dev supportsFamily:MTLGPUFamilyApple4])
726 caps.maxThreadGroupSize = 1024;
727 if ([d->dev supportsFamily:MTLGPUFamilyApple5])
728 caps.multiView = true;
729#endif
730
731 // MTLBlitCommandEncoder's buffer-to-buffer copy requires 4 byte aligned
732 // offsets and size on non-Apple GPUs, whereas uploadStaticBuffer() takes
733 // any offset and size. Keep the legacy host-write path there.
734 caps.usePrivateStaticBuffers = caps.isAppleGPU;
735 if (qEnvironmentVariableIntValue("QT_METAL_NO_PRIVATE_STATIC_BUFFERS"))
736 caps.usePrivateStaticBuffers = false;
737
738 caps.supportedSampleCounts = { 1 };
739 for (int sampleCount : { 2, 4, 8 }) {
740 if ([d->dev supportsTextureSampleCount: sampleCount])
741 caps.supportedSampleCounts.append(sampleCount);
742 }
743
744 caps.indirectCommandBuffers = ([d->dev supportsFamily:MTLGPUFamilyApple5]
745 || [d->dev supportsFamily:MTLGPUFamilyMac2])
746 && [d->dev supportsFamily:MTLGPUFamilyMetal3];
747
748 caps.shadingRateMap = [d->dev supportsRasterizationRateMapWithLayerCount: 1];
749 if (caps.shadingRateMap && caps.multiView)
750 caps.shadingRateMap = [d->dev supportsRasterizationRateMapWithLayerCount: 2];
751
752 // QTBUG-144444: setDepthClipMode is not available on the Simulator
753 caps.depthClamp = [d->dev supportsFamily:MTLGPUFamilyApple3];
754
755 if (rhiFlags.testFlag(QRhi::EnablePipelineCacheDataSave))
756 d->setupBinaryArchive();
757
758 nativeHandlesStruct.dev = (MTLDevice *) d->dev;
759 nativeHandlesStruct.cmdQueue = (MTLCommandQueue *) d->cmdQueue;
760
761 return true;
762}
763
765{
768
769 for (QMetalShader &s : d->shaderCache)
770 s.destroy();
771 d->shaderCache.clear();
772
773 [d->captureScope release];
774 d->captureScope = nil;
775
776 for (auto *pools : { d->argBufPool, d->bufStagingPool }) {
777 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
778 [pools[i].buf release];
779 pools[i].buf = nil;
780 pools[i].capacity = 0;
781 pools[i].offset = 0;
782 }
783 }
784
785 [d->icbArgumentBuffer release];
786 d->icbArgumentBuffer = nil;
787
788 [d->icbRangeBuffer release];
789 d->icbRangeBuffer = nil;
790
791 [d->icbNoCountBuffer release];
792 d->icbNoCountBuffer = nil;
793
794 [d->icbEncodeFunction release];
795 d->icbEncodeFunction = nil;
796
797 [d->icbEncodeFunctionU32 release];
798 d->icbEncodeFunctionU32 = nil;
799
800 [d->icbEncodeFunctionU16 release];
801 d->icbEncodeFunctionU16 = nil;
802
803 [d->icbEncodePipeline release];
804 d->icbEncodePipeline = nil;
805
806 [d->icbEncodePipelineU32 release];
807 d->icbEncodePipelineU32 = nil;
808
809 [d->icbEncodePipelineU16 release];
810 d->icbEncodePipelineU16 = nil;
811
812 [d->icb release];
813 d->icb = nil;
814
815 d->icbCapacity = 0;
816 d->icbSetupFailed = false;
817
818 [d->binArch release];
819 d->binArch = nil;
820
821 [d->cmdQueue release];
822 if (!importedCmdQueue)
823 d->cmdQueue = nil;
824
825 [d->dev release];
826 if (!importedDevice)
827 d->dev = nil;
828}
829
831{
832 return caps.supportedSampleCounts;
833}
834
836{
837 Q_UNUSED(sampleCount);
838 return { QSize(1, 1) };
839}
840
841QRhiSwapChain *QRhiMetal::createSwapChain()
842{
843 return new QMetalSwapChain(this);
844}
845
846QRhiBuffer *QRhiMetal::createBuffer(QRhiBuffer::Type type, QRhiBuffer::UsageFlags usage, quint32 size)
847{
848 return new QMetalBuffer(this, type, usage, size);
849}
850
852{
853 return 256;
854}
855
857{
858 return false;
859}
860
862{
863 return true;
864}
865
867{
868 return true;
869}
870
872{
873 // depth range 0..1
874 // NB the ctor takes row-major
875 static constexpr QMatrix4x4 m(1.0f, 0.0f, 0.0f, 0.0f,
876 0.0f, 1.0f, 0.0f, 0.0f,
877 0.0f, 0.0f, 0.5f, 0.5f,
878 0.0f, 0.0f, 0.0f, 1.0f);
879 return m;
880}
881
882bool QRhiMetal::isTextureFormatSupported(QRhiTexture::Format format, QRhiTexture::Flags flags) const
883{
884 Q_UNUSED(flags);
885
886 bool supportsFamilyMac2 = false; // needed for BC* formats
887 bool supportsFamilyApple3 = false;
888
889#ifdef Q_OS_MACOS
890 supportsFamilyMac2 = true;
891 if (caps.isAppleGPU)
892 supportsFamilyApple3 = true;
893#else
894 supportsFamilyApple3 = true;
895#endif
896
897 // BC5 is not available for any Apple hardare
898 if (format == QRhiTexture::BC5)
899 return false;
900
901 if (!supportsFamilyApple3) {
902 if (format >= QRhiTexture::ETC2_RGB8 && format <= QRhiTexture::ETC2_RGBA8)
903 return false;
904 if (format >= QRhiTexture::ASTC_4x4 && format <= QRhiTexture::ASTC_12x12)
905 return false;
906 }
907
908 if (!supportsFamilyMac2)
909 if (format >= QRhiTexture::BC1 && format <= QRhiTexture::BC7)
910 return false;
911
912 return true;
913}
914
915bool QRhiMetal::isFeatureSupported(QRhi::Feature feature) const
916{
917 switch (feature) {
918 case QRhi::MultisampleTexture:
919 return true;
920 case QRhi::MultisampleRenderBuffer:
921 return true;
922 case QRhi::DebugMarkers:
923 return true;
924 case QRhi::Timestamps:
925 return true;
926 case QRhi::Instancing:
927 return true;
928 case QRhi::CustomInstanceStepRate:
929 return true;
930 case QRhi::PrimitiveRestart:
931 return true;
932 case QRhi::NonDynamicUniformBuffers:
933 return true;
934 case QRhi::NonFourAlignedEffectiveIndexBufferOffset:
935 return false;
936 case QRhi::NPOTTextureRepeat:
937 return true;
938 case QRhi::RedOrAlpha8IsRed:
939 return true;
940 case QRhi::ElementIndexUint:
941 return true;
942 case QRhi::Compute:
943 return true;
944 case QRhi::WideLines:
945 return false;
946 case QRhi::VertexShaderPointSize:
947 return true;
948 case QRhi::BaseVertex:
949 return caps.baseVertexAndInstance;
950 case QRhi::BaseInstance:
951 return caps.baseVertexAndInstance;
952 case QRhi::TriangleFanTopology:
953 return false;
954 case QRhi::ReadBackNonUniformBuffer:
955 return true;
956 case QRhi::ReadBackNonBaseMipLevel:
957 return true;
958 case QRhi::TexelFetch:
959 return true;
960 case QRhi::RenderToNonBaseMipLevel:
961 return true;
962 case QRhi::IntAttributes:
963 return true;
964 case QRhi::ScreenSpaceDerivatives:
965 return true;
966 case QRhi::ReadBackAnyTextureFormat:
967 return true;
968 case QRhi::PipelineCacheDataLoadSave:
970 return false;
971#else
972 return true;
973#endif
974 case QRhi::ImageDataStride:
975 return true;
976 case QRhi::RenderBufferImport:
977 return false;
978 case QRhi::ThreeDimensionalTextures:
979 return true;
980 case QRhi::RenderTo3DTextureSlice:
981 return true;
982 case QRhi::TextureArrays:
983 return true;
984 case QRhi::Tessellation:
985 return true;
986 case QRhi::GeometryShader:
987 return false;
988 case QRhi::TextureArrayRange:
989 return true;
990 case QRhi::NonFillPolygonMode:
991 return true;
992 case QRhi::OneDimensionalTextures:
993 return true;
994 case QRhi::OneDimensionalTextureMipmaps:
995 return false;
996 case QRhi::HalfAttributes:
997 return true;
998 case QRhi::RenderToOneDimensionalTexture:
999 return false;
1000 case QRhi::ThreeDimensionalTextureMipmaps:
1001 return true;
1002 case QRhi::MultiView:
1003 return caps.multiView;
1004 case QRhi::TextureViewFormat:
1005 return true;
1006 case QRhi::ResolveDepthStencil:
1007 return true;
1008 case QRhi::VariableRateShading:
1009 return false;
1010 case QRhi::VariableRateShadingMap:
1011 return caps.shadingRateMap;
1012 case QRhi::VariableRateShadingMapWithTexture:
1013 return false;
1014 case QRhi::PerRenderTargetBlending:
1015 case QRhi::SampleVariables:
1016 return true;
1017 case QRhi::InstanceIndexIncludesBaseInstance:
1018 return true;
1019 case QRhi::DepthClamp:
1020 return caps.depthClamp;
1021 case QRhi::DrawIndirect:
1022 return true;
1023 case QRhi::DrawIndirectMulti:
1024 return caps.indirectCommandBuffers;
1025 case QRhi::ShaderDrawParameters:
1026 return false;
1027 case QRhi::PushConstants:
1028 return true;
1029 case QRhi::DrawIndirectCount:
1030 return caps.indirectCommandBuffers;
1031 case QRhi::DispatchIndirect:
1032 return true;
1033 case QRhi::BufferToBufferCopy:
1034 return true;
1035 case QRhi::StaticBuffersOnGpuTimeline:
1036 return caps.usePrivateStaticBuffers;
1037 default:
1038 Q_UNREACHABLE();
1039 return false;
1040 }
1041}
1042
1043int QRhiMetal::resourceLimit(QRhi::ResourceLimit limit) const
1044{
1045 switch (limit) {
1046 case QRhi::TextureSizeMin:
1047 return 1;
1048 case QRhi::TextureSizeMax:
1049 return caps.maxTextureSize;
1050 case QRhi::MaxColorAttachments:
1051 return 8;
1052 case QRhi::FramesInFlight:
1053 return QMTL_FRAMES_IN_FLIGHT;
1054 case QRhi::MaxAsyncReadbackFrames:
1055 return QMTL_FRAMES_IN_FLIGHT;
1056 case QRhi::MaxThreadGroupsPerDimension:
1057 return 65535;
1058 case QRhi::MaxThreadsPerThreadGroup:
1059 Q_FALLTHROUGH();
1060 case QRhi::MaxThreadGroupX:
1061 Q_FALLTHROUGH();
1062 case QRhi::MaxThreadGroupY:
1063 Q_FALLTHROUGH();
1064 case QRhi::MaxThreadGroupZ:
1065 return caps.maxThreadGroupSize;
1066 case QRhi::TextureArraySizeMax:
1067 return 2048;
1068 case QRhi::MaxUniformBufferRange:
1069 return 65536;
1070 case QRhi::MaxVertexInputs:
1071 return 31;
1072 case QRhi::MaxVertexOutputs:
1073 return 15; // use the minimum from MTLGPUFamily1/2/3
1074 case QRhi::MaxPushConstantsSize:
1075 return 4096; // the limit for setVertexBytes() and friends
1076 case QRhi::MaxVertexStorageBuffers:
1077 case QRhi::MaxFragmentStorageBuffers:
1078 return 31;
1079 case QRhi::ShadingRateImageTileSize:
1080 return 0;
1081 default:
1082 Q_UNREACHABLE();
1083 return 0;
1084 }
1085}
1086
1088{
1089 return &nativeHandlesStruct;
1090}
1091
1093{
1094 return driverInfoStruct;
1095}
1096
1098{
1099 QRhiStats result;
1100 result.totalPipelineCreationTime = totalPipelineCreationTime();
1101 return result;
1102}
1103
1105{
1106 // not applicable
1107 return false;
1108}
1109
1110void QRhiMetal::setQueueSubmitParams(QRhiNativeHandles *)
1111{
1112 // not applicable
1113}
1114
1116{
1117 for (QMetalShader &s : d->shaderCache)
1118 s.destroy();
1119
1120 d->shaderCache.clear();
1121}
1122
1124{
1125 return false;
1126}
1127
1137
1139{
1140 Q_STATIC_ASSERT(sizeof(QMetalPipelineCacheDataHeader) == 256);
1141 QByteArray data;
1142 if (!d->binArch || !rhiFlags.testFlag(QRhi::EnablePipelineCacheDataSave))
1143 return data;
1144
1145 QTemporaryFile tmp;
1146 if (!tmp.open()) {
1147 qCDebug(QRHI_LOG_INFO, "pipelineCacheData: Failed to create temporary file for Metal");
1148 return data;
1149 }
1150 tmp.close(); // the file exists until the tmp dtor runs
1151
1152 const QString fn = QFileInfo(tmp.fileName()).absoluteFilePath();
1153 NSURL *url = QUrl::fromLocalFile(fn).toNSURL();
1154 NSError *err = nil;
1155 if (![d->binArch serializeToURL: url error: &err]) {
1156 const QString msg = QString::fromNSString(err.localizedDescription);
1157 // Some of these "errors" are not actual errors. (think of "Nothing to serialize")
1158 qCDebug(QRHI_LOG_INFO, "Failed to serialize MTLBinaryArchive: %s", qPrintable(msg));
1159 return data;
1160 }
1161
1162 QFile f(fn);
1163 if (!f.open(QIODevice::ReadOnly)) {
1164 qCDebug(QRHI_LOG_INFO, "pipelineCacheData: Failed to reopen temporary file");
1165 return data;
1166 }
1167 const QByteArray blob = f.readAll();
1168 f.close();
1169
1170 const size_t headerSize = sizeof(QMetalPipelineCacheDataHeader);
1171 const quint32 dataSize = quint32(blob.size());
1172
1173 data.resize(headerSize + dataSize);
1174
1176 header.rhiId = pipelineCacheRhiId();
1177 header.arch = quint32(sizeof(void*));
1178 header.dataSize = quint32(dataSize);
1179 header.osMajor = osMajor;
1180 header.osMinor = osMinor;
1181 const size_t driverStrLen = qMin(sizeof(header.driver) - 1, size_t(driverInfoStruct.deviceName.length()));
1182 if (driverStrLen)
1183 memcpy(header.driver, driverInfoStruct.deviceName.constData(), driverStrLen);
1184 header.driver[driverStrLen] = '\0';
1185
1186 memcpy(data.data(), &header, headerSize);
1187 memcpy(data.data() + headerSize, blob.constData(), dataSize);
1188 return data;
1189}
1190
1191void QRhiMetal::setPipelineCacheData(const QByteArray &data)
1192{
1193 if (data.isEmpty())
1194 return;
1195
1196 const size_t headerSize = sizeof(QMetalPipelineCacheDataHeader);
1197 if (data.size() < qsizetype(headerSize)) {
1198 qCDebug(QRHI_LOG_INFO, "setPipelineCacheData: Invalid blob size (header incomplete)");
1199 return;
1200 }
1201
1202 const size_t dataOffset = headerSize;
1204 memcpy(&header, data.constData(), headerSize);
1205
1206 const quint32 rhiId = pipelineCacheRhiId();
1207 if (header.rhiId != rhiId) {
1208 qCDebug(QRHI_LOG_INFO, "setPipelineCacheData: The data is for a different QRhi version or backend (%u, %u)",
1209 rhiId, header.rhiId);
1210 return;
1211 }
1212
1213 const quint32 arch = quint32(sizeof(void*));
1214 if (header.arch != arch) {
1215 qCDebug(QRHI_LOG_INFO, "setPipelineCacheData: Architecture does not match (%u, %u)",
1216 arch, header.arch);
1217 return;
1218 }
1219
1220 if (header.osMajor != osMajor || header.osMinor != osMinor) {
1221 qCDebug(QRHI_LOG_INFO, "setPipelineCacheData: OS version does not match (%u.%u, %u.%u)",
1222 osMajor, osMinor, header.osMajor, header.osMinor);
1223 return;
1224 }
1225
1226 const size_t driverStrLen = qMin(sizeof(header.driver) - 1, size_t(driverInfoStruct.deviceName.length()));
1227 if (strncmp(header.driver, driverInfoStruct.deviceName.constData(), driverStrLen)) {
1228 qCDebug(QRHI_LOG_INFO, "setPipelineCacheData: Metal device name does not match");
1229 return;
1230 }
1231
1232 if (quint64(data.size()) < quint64(dataOffset) + header.dataSize) {
1233 qCDebug(QRHI_LOG_INFO, "setPipelineCacheData: Invalid blob size (data incomplete)");
1234 return;
1235 }
1236
1237 const char *p = data.constData() + dataOffset;
1238
1239 QTemporaryFile tmp;
1240 if (!tmp.open()) {
1241 qCDebug(QRHI_LOG_INFO, "pipelineCacheData: Failed to create temporary file for Metal");
1242 return;
1243 }
1244 tmp.write(p, header.dataSize);
1245 tmp.close(); // the file exists until the tmp dtor runs
1246
1247 const QString fn = QFileInfo(tmp.fileName()).absoluteFilePath();
1248 NSURL *url = QUrl::fromLocalFile(fn).toNSURL();
1249 if (d->setupBinaryArchive(url))
1250 qCDebug(QRHI_LOG_INFO, "Created MTLBinaryArchive with initial data of %u bytes", header.dataSize);
1251}
1252
1253QRhiRenderBuffer *QRhiMetal::createRenderBuffer(QRhiRenderBuffer::Type type, const QSize &pixelSize,
1254 int sampleCount, QRhiRenderBuffer::Flags flags,
1255 QRhiTexture::Format backingFormatHint)
1256{
1257 return new QMetalRenderBuffer(this, type, pixelSize, sampleCount, flags, backingFormatHint);
1258}
1259
1260QRhiTexture *QRhiMetal::createTexture(QRhiTexture::Format format,
1261 const QSize &pixelSize, int depth, int arraySize,
1262 int sampleCount, QRhiTexture::Flags flags)
1263{
1264 return new QMetalTexture(this, format, pixelSize, depth, arraySize, sampleCount, flags);
1265}
1266
1267QRhiSampler *QRhiMetal::createSampler(QRhiSampler::Filter magFilter, QRhiSampler::Filter minFilter,
1268 QRhiSampler::Filter mipmapMode,
1269 QRhiSampler::AddressMode u, QRhiSampler::AddressMode v, QRhiSampler::AddressMode w)
1270{
1271 return new QMetalSampler(this, magFilter, minFilter, mipmapMode, u, v, w);
1272}
1273
1274QRhiShadingRateMap *QRhiMetal::createShadingRateMap()
1275{
1276 return new QMetalShadingRateMap(this);
1277}
1278
1279QRhiTextureRenderTarget *QRhiMetal::createTextureRenderTarget(const QRhiTextureRenderTargetDescription &desc,
1280 QRhiTextureRenderTarget::Flags flags)
1281{
1282 return new QMetalTextureRenderTarget(this, desc, flags);
1283}
1284
1286{
1287 return new QMetalGraphicsPipeline(this);
1288}
1289
1291{
1292 return new QMetalComputePipeline(this);
1293}
1294
1296{
1297 return new QMetalShaderResourceBindings(this);
1298}
1299
1305
1306static inline int mapBinding(int binding,
1307 int stageIndex,
1308 const QShader::NativeResourceBindingMap *nativeResourceBindingMaps[],
1309 BindingType type)
1310{
1311 const QShader::NativeResourceBindingMap *map = nativeResourceBindingMaps[stageIndex];
1312 if (!map || map->isEmpty())
1313 return binding; // old QShader versions do not have this map, assume 1:1 mapping then
1314
1315 auto it = map->constFind(binding);
1316 if (it != map->cend())
1317 return type == BindingType::Sampler ? it->second : it->first; // may be -1, if the resource is inactive
1318
1319 // Hitting this path is normal too. It is not given that the resource (for
1320 // example, a uniform block) is present in the shaders for all the stages
1321 // specified by the visibility mask in the QRhiShaderResourceBinding.
1322 return -1;
1323}
1324
1325static inline MTLResourceUsage storageImageUsage(QRhiShaderResourceBinding::Type type)
1326{
1327 switch (type) {
1328 case QRhiShaderResourceBinding::ImageLoad:
1329 return MTLResourceUsageRead;
1330 case QRhiShaderResourceBinding::ImageStore:
1331 return MTLResourceUsageWrite;
1332 default:
1333 return MTLResourceUsageRead | MTLResourceUsageWrite;
1334 }
1335}
1336
1338 int encoderStage,
1340{
1341 for (const QMetalShaderResourceBindingsData::Stage::Texture &t : res.textures) {
1342 switch (encoderStage) {
1343 case QMetalShaderResourceBindingsData::VERTEX:
1344 [cbD->d->currentRenderPassEncoder useResource: t.mtltex usage: t.usage stages: MTLRenderStageVertex];
1345 break;
1346 case QMetalShaderResourceBindingsData::FRAGMENT:
1347 [cbD->d->currentRenderPassEncoder useResource: t.mtltex usage: t.usage stages: MTLRenderStageFragment];
1348 break;
1349 case QMetalShaderResourceBindingsData::COMPUTE:
1350 [cbD->d->currentComputePassEncoder useResource: t.mtltex usage: t.usage];
1351 break;
1352 default:
1353 break;
1354 }
1355 }
1356}
1357
1359 int stage,
1360 const QRhiBatchedBindings<id<MTLBuffer>>::Batch &bufferBatch,
1361 const QRhiBatchedBindings<NSUInteger>::Batch &offsetBatch)
1362{
1363 switch (stage) {
1364 case QMetalShaderResourceBindingsData::VERTEX:
1365 [cbD->d->currentRenderPassEncoder setVertexBuffers: bufferBatch.resources.constData()
1366 offsets: offsetBatch.resources.constData()
1367 withRange: NSMakeRange(bufferBatch.startBinding, NSUInteger(bufferBatch.resources.count()))];
1368 break;
1369 case QMetalShaderResourceBindingsData::FRAGMENT:
1370 [cbD->d->currentRenderPassEncoder setFragmentBuffers: bufferBatch.resources.constData()
1371 offsets: offsetBatch.resources.constData()
1372 withRange: NSMakeRange(bufferBatch.startBinding, NSUInteger(bufferBatch.resources.count()))];
1373 break;
1374 case QMetalShaderResourceBindingsData::COMPUTE:
1375 [cbD->d->currentComputePassEncoder setBuffers: bufferBatch.resources.constData()
1376 offsets: offsetBatch.resources.constData()
1377 withRange: NSMakeRange(bufferBatch.startBinding, NSUInteger(bufferBatch.resources.count()))];
1378 break;
1381 // do nothing. These are used later for tessellation
1382 break;
1383 default:
1384 Q_UNREACHABLE();
1385 break;
1386 }
1387}
1388
1390 int stage,
1391 const QRhiBatchedBindings<id<MTLTexture>>::Batch &textureBatch)
1392{
1393 switch (stage) {
1394 case QMetalShaderResourceBindingsData::VERTEX:
1395 [cbD->d->currentRenderPassEncoder setVertexTextures: textureBatch.resources.constData()
1396 withRange: NSMakeRange(textureBatch.startBinding, NSUInteger(textureBatch.resources.count()))];
1397 break;
1398 case QMetalShaderResourceBindingsData::FRAGMENT:
1399 [cbD->d->currentRenderPassEncoder setFragmentTextures: textureBatch.resources.constData()
1400 withRange: NSMakeRange(textureBatch.startBinding, NSUInteger(textureBatch.resources.count()))];
1401 break;
1402 case QMetalShaderResourceBindingsData::COMPUTE:
1403 [cbD->d->currentComputePassEncoder setTextures: textureBatch.resources.constData()
1404 withRange: NSMakeRange(textureBatch.startBinding, NSUInteger(textureBatch.resources.count()))];
1405 break;
1408 // do nothing. These are used later for tessellation
1409 break;
1410 default:
1411 Q_UNREACHABLE();
1412 break;
1413 }
1414}
1415
1417 int encoderStage,
1418 const QRhiBatchedBindings<id<MTLSamplerState>>::Batch &samplerBatch)
1419{
1420 switch (encoderStage) {
1421 case QMetalShaderResourceBindingsData::VERTEX:
1422 [cbD->d->currentRenderPassEncoder setVertexSamplerStates: samplerBatch.resources.constData()
1423 withRange: NSMakeRange(samplerBatch.startBinding, NSUInteger(samplerBatch.resources.count()))];
1424 break;
1425 case QMetalShaderResourceBindingsData::FRAGMENT:
1426 [cbD->d->currentRenderPassEncoder setFragmentSamplerStates: samplerBatch.resources.constData()
1427 withRange: NSMakeRange(samplerBatch.startBinding, NSUInteger(samplerBatch.resources.count()))];
1428 break;
1429 case QMetalShaderResourceBindingsData::COMPUTE:
1430 [cbD->d->currentComputePassEncoder setSamplerStates: samplerBatch.resources.constData()
1431 withRange: NSMakeRange(samplerBatch.startBinding, NSUInteger(samplerBatch.resources.count()))];
1432 break;
1435 // do nothing. These are used later for tessellation
1436 break;
1437 default:
1438 Q_UNREACHABLE();
1439 break;
1440 }
1441}
1442
1443// Helper that is not used during the common vertex+fragment and compute
1444// pipelines, but is necessary when tessellation is involved and so the
1445// graphics pipeline is under the hood a combination of multiple compute and
1446// render pipelines. We need to be able to set the buffers, textures, samplers
1447// when a switching between render and compute encoders.
1448static inline void rebindShaderResources(QMetalCommandBuffer *cbD, int resourceStage, int encoderStage,
1449 const QMetalShaderResourceBindingsData *customBindingState = nullptr)
1450{
1451 const QMetalShaderResourceBindingsData *bindingData = customBindingState ? customBindingState : &cbD->d->currentShaderResourceBindingState;
1452
1453 for (int i = 0, ie = bindingData->res[resourceStage].bufferBatches.batches.count(); i != ie; ++i) {
1454 const auto &bufferBatch(bindingData->res[resourceStage].bufferBatches.batches[i]);
1455 const auto &offsetBatch(bindingData->res[resourceStage].bufferOffsetBatches.batches[i]);
1456 bindStageBuffers(cbD, encoderStage, bufferBatch, offsetBatch);
1457 }
1458
1459 for (int i = 0, ie = bindingData->res[resourceStage].textureBatches.batches.count(); i != ie; ++i) {
1460 const auto &batch(bindingData->res[resourceStage].textureBatches.batches[i]);
1461 bindStageTextures(cbD, encoderStage, batch);
1462 }
1463
1464 for (int i = 0, ie = bindingData->res[resourceStage].samplerBatches.batches.count(); i != ie; ++i) {
1465 const auto &batch(bindingData->res[resourceStage].samplerBatches.batches[i]);
1466 bindStageSamplers(cbD, encoderStage, batch);
1467 }
1468
1469 if (bindingData->res[resourceStage].usesArgumentBuffer)
1470 declareStageArgumentBufferResources(cbD, encoderStage, bindingData->res[resourceStage]);
1471}
1472
1474{
1475 switch (stage) {
1476 case QMetalShaderResourceBindingsData::VERTEX:
1477 return QRhiShaderResourceBinding::StageFlag::VertexStage;
1478 case QMetalShaderResourceBindingsData::TESSCTRL:
1479 return QRhiShaderResourceBinding::StageFlag::TessellationControlStage;
1480 case QMetalShaderResourceBindingsData::TESSEVAL:
1481 return QRhiShaderResourceBinding::StageFlag::TessellationEvaluationStage;
1482 case QMetalShaderResourceBindingsData::FRAGMENT:
1483 return QRhiShaderResourceBinding::StageFlag::FragmentStage;
1484 case QMetalShaderResourceBindingsData::COMPUTE:
1485 return QRhiShaderResourceBinding::StageFlag::ComputeStage;
1486 }
1487
1488 Q_UNREACHABLE_RETURN(QRhiShaderResourceBinding::StageFlag::VertexStage);
1489}
1490
1493 int dynamicOffsetCount,
1494 const QRhiCommandBuffer::DynamicOffset *dynamicOffsets,
1495 bool offsetOnlyChange,
1496 const QShader::NativeResourceBindingMap *nativeResourceBindingMaps[SUPPORTED_STAGES],
1497 const QMetalShader *shaders[SUPPORTED_STAGES])
1498{
1500
1501 for (const QRhiShaderResourceBinding &binding : std::as_const(srbD->sortedBindings)) {
1502 const QRhiShaderResourceBinding::Data *b = shaderResourceBindingData(binding);
1503 switch (b->type) {
1504 case QRhiShaderResourceBinding::UniformBuffer:
1505 {
1506 QMetalBuffer *bufD = QRHI_RES(QMetalBuffer, b->u.ubuf.buf);
1507 id<MTLBuffer> mtlbuf = bufD->d->buf[bufD->d->slotted ? currentFrameSlot : 0];
1508 quint32 offset = b->u.ubuf.offset;
1509 for (int i = 0; i < dynamicOffsetCount; ++i) {
1510 const QRhiCommandBuffer::DynamicOffset &dynOfs(dynamicOffsets[i]);
1511 if (dynOfs.first == b->binding) {
1512 offset = dynOfs.second;
1513 break;
1514 }
1515 }
1516
1517 for (int stage = 0; stage < SUPPORTED_STAGES; ++stage) {
1518 if (b->stage.testFlag(toRhiSrbStage(stage))) {
1519 const int nativeBinding = mapBinding(b->binding, stage, nativeResourceBindingMaps, BindingType::Buffer);
1520 if (nativeBinding >= 0)
1521 bindingData.res[stage].buffers.append({ nativeBinding, mtlbuf, offset });
1522 }
1523 }
1524 }
1525 break;
1526 case QRhiShaderResourceBinding::SampledTexture:
1527 case QRhiShaderResourceBinding::Texture:
1528 case QRhiShaderResourceBinding::Sampler:
1529 {
1530 const QRhiShaderResourceBinding::Data::TextureAndOrSamplerData *data = &b->stex;
1531 for (int elem = 0; elem < data->count(); ++elem) {
1532 QMetalTexture *texD = QRHI_RES(QMetalTexture, b->stex.texSamplers[elem].tex);
1533 QMetalSampler *samplerD = QRHI_RES(QMetalSampler, b->stex.texSamplers[elem].sampler);
1534
1535 for (int stage = 0; stage < SUPPORTED_STAGES; ++stage) {
1536 if (b->stage.testFlag(toRhiSrbStage(stage))) {
1537 // Must handle all three cases (combined, separate, separate):
1538 // first = texture binding, second = sampler binding
1539 // first = texture binding
1540 // first = sampler binding (i.e. BindingType::Texture...)
1541 const int textureBinding = mapBinding(b->binding, stage, nativeResourceBindingMaps, BindingType::Texture);
1542 const int samplerBinding = texD && samplerD ? mapBinding(b->binding, stage, nativeResourceBindingMaps, BindingType::Sampler)
1543 : (samplerD ? mapBinding(b->binding, stage, nativeResourceBindingMaps, BindingType::Texture) : -1);
1544 if (textureBinding >= 0 && texD)
1545 bindingData.res[stage].textures.append({ textureBinding + elem, texD->d->textureForSampling(), MTLResourceUsageRead });
1546 if (samplerBinding >= 0)
1547 bindingData.res[stage].samplers.append({ samplerBinding + elem, samplerD->d->samplerState });
1548 }
1549 }
1550 }
1551 }
1552 break;
1553 case QRhiShaderResourceBinding::ImageLoad:
1554 case QRhiShaderResourceBinding::ImageStore:
1555 case QRhiShaderResourceBinding::ImageLoadStore:
1556 {
1557 QMetalTexture *texD = QRHI_RES(QMetalTexture, b->u.simage.tex);
1558 id<MTLTexture> t = texD->d->viewForLevel(b->u.simage.level);
1559
1560 for (int stage = 0; stage < SUPPORTED_STAGES; ++stage) {
1561 if (b->stage.testFlag(toRhiSrbStage(stage))) {
1562 const int nativeBinding = mapBinding(b->binding, stage, nativeResourceBindingMaps, BindingType::Texture);
1563 if (nativeBinding >= 0)
1564 bindingData.res[stage].textures.append({ nativeBinding, t, storageImageUsage(b->type) });
1565 }
1566 }
1567 }
1568 break;
1569 case QRhiShaderResourceBinding::BufferLoad:
1570 case QRhiShaderResourceBinding::BufferStore:
1571 case QRhiShaderResourceBinding::BufferLoadStore:
1572 {
1573 QMetalBuffer *bufD = QRHI_RES(QMetalBuffer, b->u.sbuf.buf);
1574 id<MTLBuffer> mtlbuf = bufD->d->buf[bufD->d->slotted ? currentFrameSlot : 0];
1575 quint32 offset = b->u.sbuf.offset;
1576 for (int stage = 0; stage < SUPPORTED_STAGES; ++stage) {
1577 if (b->stage.testFlag(toRhiSrbStage(stage))) {
1578 const int nativeBinding = mapBinding(b->binding, stage, nativeResourceBindingMaps, BindingType::Buffer);
1579 if (nativeBinding >= 0)
1580 bindingData.res[stage].buffers.append({ nativeBinding, mtlbuf, offset });
1581 }
1582 }
1583 }
1584 break;
1585 default:
1586 Q_UNREACHABLE();
1587 break;
1588 }
1589 }
1590
1591 // With the argument buffer shader variant the texture and sampler values
1592 // collected above are ids within the argument buffer, so encode them into
1593 // one and bind that instead of binding them individually.
1594 for (int stage = 0; stage < SUPPORTED_STAGES; ++stage) {
1595 const QMetalShader *shader = shaders[stage];
1596 if (!shader || !shader->argumentEncoder)
1597 continue;
1598 QMetalShaderResourceBindingsData::Stage &res(bindingData.res[stage]);
1599
1600 // offsetOnlyChange means same srb, same pipeline, and no resource
1601 // changed, so what is in the argument buffer cannot have changed
1602 // either. Carry the encoded one over instead of building it again.
1603 if (offsetOnlyChange) {
1605 cbD->d->currentShaderResourceBindingState.res[stage]);
1606 if (prev.usesArgumentBuffer) {
1607 for (const QMetalShaderResourceBindingsData::Stage::Buffer &b : prev.buffers) {
1608 if (b.nativeBinding == shader->argumentBufferIndex) {
1609 res.samplers.clear();
1610 res.buffers.append(b);
1611 res.usesArgumentBuffer = true;
1612 break;
1613 }
1614 }
1615 if (res.usesArgumentBuffer)
1616 continue;
1617 }
1618 }
1619
1620 // The argument buffer is in the constant address space, and it is bound
1621 // like any other buffer, so the offset must satisfy the same 256 byte
1622 // requirement on macOS that makes ubufAlignment() 256. The encoder's
1623 // own alignment is typically well below that.
1624 const quint32 argBufAlignment = qMax(quint32(shader->argumentEncoder.alignment),
1625 quint32(ubufAlignment()));
1626 quint32 argBufOffset = 0;
1627 id<MTLBuffer> argBuf = d->allocArgumentBuffer(quint32(shader->argumentEncoder.encodedLength),
1628 argBufAlignment, currentFrameSlot, &argBufOffset);
1629 if (!argBuf) {
1630 // Nothing gets bound at the argument buffer's index then, so the
1631 // shader will dereference garbage. Acceptable: failing to
1632 // sub-allocate from a small shared memory buffer means the process
1633 // is out of memory anyway.
1634 qWarning("Failed to allocate Metal argument buffer");
1635 res.textures.clear();
1636 res.samplers.clear();
1637 continue;
1638 }
1639 [shader->argumentEncoder setArgumentBuffer: argBuf offset: argBufOffset];
1640 for (const QMetalShaderResourceBindingsData::Stage::Texture &t : std::as_const(res.textures))
1641 [shader->argumentEncoder setTexture: t.mtltex atIndex: NSUInteger(t.nativeBinding)];
1642 for (const QMetalShaderResourceBindingsData::Stage::Sampler &sm : std::as_const(res.samplers))
1643 [shader->argumentEncoder setSamplerState: sm.mtlsampler atIndex: NSUInteger(sm.nativeBinding)];
1644 res.samplers.clear();
1645 res.buffers.append({ shader->argumentBufferIndex, argBuf, argBufOffset });
1646 res.usesArgumentBuffer = true;
1647 }
1648
1649 for (int stage = 0; stage < SUPPORTED_STAGES; ++stage) {
1652 continue;
1654 continue;
1655
1656 // QRhiBatchedBindings works with the native bindings and expects
1657 // sorted input. The pre-sorted QRhiShaderResourceBinding list (based
1658 // on the QRhi (SPIR-V) binding) is not helpful in this regard, so we
1659 // have to sort here every time.
1660
1661 std::sort(bindingData.res[stage].buffers.begin(), bindingData.res[stage].buffers.end(), [](const QMetalShaderResourceBindingsData::Stage::Buffer &a, const QMetalShaderResourceBindingsData::Stage::Buffer &b) {
1662 return a.nativeBinding < b.nativeBinding;
1663 });
1664
1665 for (const QMetalShaderResourceBindingsData::Stage::Buffer &buf : std::as_const(bindingData.res[stage].buffers)) {
1666 bindingData.res[stage].bufferBatches.feed(buf.nativeBinding, buf.mtlbuf);
1667 bindingData.res[stage].bufferOffsetBatches.feed(buf.nativeBinding, buf.offset);
1668 }
1669
1670 bindingData.res[stage].bufferBatches.finish();
1671 bindingData.res[stage].bufferOffsetBatches.finish();
1672
1673 for (int i = 0, ie = bindingData.res[stage].bufferBatches.batches.count(); i != ie; ++i) {
1674 const auto &bufferBatch(bindingData.res[stage].bufferBatches.batches[i]);
1675 const auto &offsetBatch(bindingData.res[stage].bufferOffsetBatches.batches[i]);
1676 // skip setting Buffer binding if the current state is already correct
1677 if (cbD->d->currentShaderResourceBindingState.res[stage].bufferBatches.batches.count() > i
1678 && cbD->d->currentShaderResourceBindingState.res[stage].bufferOffsetBatches.batches.count() > i
1679 && bufferBatch == cbD->d->currentShaderResourceBindingState.res[stage].bufferBatches.batches[i]
1680 && offsetBatch == cbD->d->currentShaderResourceBindingState.res[stage].bufferOffsetBatches.batches[i])
1681 {
1682 continue;
1683 }
1684 bindStageBuffers(cbD, stage, bufferBatch, offsetBatch);
1685 }
1686
1687 if (offsetOnlyChange)
1688 continue;
1689
1690 if (bindingData.res[stage].usesArgumentBuffer) {
1691 declareStageArgumentBufferResources(cbD, stage, bindingData.res[stage]);
1692 continue;
1693 }
1694
1695 std::sort(bindingData.res[stage].textures.begin(), bindingData.res[stage].textures.end(), [](const QMetalShaderResourceBindingsData::Stage::Texture &a, const QMetalShaderResourceBindingsData::Stage::Texture &b) {
1696 return a.nativeBinding < b.nativeBinding;
1697 });
1698
1699 std::sort(bindingData.res[stage].samplers.begin(), bindingData.res[stage].samplers.end(), [](const QMetalShaderResourceBindingsData::Stage::Sampler &a, const QMetalShaderResourceBindingsData::Stage::Sampler &b) {
1700 return a.nativeBinding < b.nativeBinding;
1701 });
1702
1703 for (const QMetalShaderResourceBindingsData::Stage::Texture &t : std::as_const(bindingData.res[stage].textures))
1704 bindingData.res[stage].textureBatches.feed(t.nativeBinding, t.mtltex);
1705
1706 for (const QMetalShaderResourceBindingsData::Stage::Sampler &s : std::as_const(bindingData.res[stage].samplers))
1707 bindingData.res[stage].samplerBatches.feed(s.nativeBinding, s.mtlsampler);
1708
1709 bindingData.res[stage].textureBatches.finish();
1710 bindingData.res[stage].samplerBatches.finish();
1711
1712 for (int i = 0, ie = bindingData.res[stage].textureBatches.batches.count(); i != ie; ++i) {
1713 const auto &batch(bindingData.res[stage].textureBatches.batches[i]);
1714 // skip setting Texture binding if the current state is already correct
1715 if (cbD->d->currentShaderResourceBindingState.res[stage].textureBatches.batches.count() > i
1716 && batch == cbD->d->currentShaderResourceBindingState.res[stage].textureBatches.batches[i])
1717 {
1718 continue;
1719 }
1720 bindStageTextures(cbD, stage, batch);
1721 }
1722
1723 for (int i = 0, ie = bindingData.res[stage].samplerBatches.batches.count(); i != ie; ++i) {
1724 const auto &batch(bindingData.res[stage].samplerBatches.batches[i]);
1725 // skip setting Sampler State if the current state is already correct
1726 if (cbD->d->currentShaderResourceBindingState.res[stage].samplerBatches.batches.count() > i
1727 && batch == cbD->d->currentShaderResourceBindingState.res[stage].samplerBatches.batches[i])
1728 {
1729 continue;
1730 }
1731 bindStageSamplers(cbD, stage, batch);
1732 }
1733 }
1734
1735 cbD->d->currentShaderResourceBindingState = bindingData;
1736}
1737
1739{
1740 const QList<QShaderDescription::PushConstantBlock> blocks = s.desc.pushConstantBlocks();
1741 return blocks.isEmpty() ? 0 : quint32(blocks.first().size);
1742}
1743
1744static inline int mtlPushConstantBufferIndex(const QMetalShader &s)
1745{
1746 return s.nativeShaderInfo.extraBufferBindings.value(QShaderPrivate::MslPushConstantBufferBinding, -1);
1747}
1748
1750{
1751 QRHI_RES_RHI(QRhiMetal);
1752
1753 // Also update the tracked state, so that the callers that reactivate a
1754 // pipeline on a new encoder after interrupting the pass do not have to.
1755 cbD->currentGraphicsPipeline = this;
1756 cbD->currentComputePipeline = nullptr;
1758
1759 [cbD->d->currentRenderPassEncoder setRenderPipelineState: d->ps];
1760
1761 if (cbD->d->currentDepthStencilState != d->ds) {
1762 [cbD->d->currentRenderPassEncoder setDepthStencilState: d->ds];
1763 cbD->d->currentDepthStencilState = d->ds;
1764 }
1765 if (cbD->currentCullMode == -1 || d->cullMode != uint(cbD->currentCullMode)) {
1766 [cbD->d->currentRenderPassEncoder setCullMode: d->cullMode];
1767 cbD->currentCullMode = int(d->cullMode);
1768 }
1769 if (cbD->currentTriangleFillMode == -1 || d->triangleFillMode != uint(cbD->currentTriangleFillMode)) {
1770 [cbD->d->currentRenderPassEncoder setTriangleFillMode: d->triangleFillMode];
1771 cbD->currentTriangleFillMode = int(d->triangleFillMode);
1772 }
1773 if (rhiD->caps.depthClamp) {
1774 if (cbD->currentDepthClipMode == -1 || d->depthClipMode != uint(cbD->currentDepthClipMode)) {
1775 [cbD->d->currentRenderPassEncoder setDepthClipMode: d->depthClipMode];
1776 cbD->currentDepthClipMode = int(d->depthClipMode);
1777 }
1778 }
1779 if (cbD->currentFrontFaceWinding == -1 || d->winding != uint(cbD->currentFrontFaceWinding)) {
1780 [cbD->d->currentRenderPassEncoder setFrontFacingWinding: d->winding];
1781 cbD->currentFrontFaceWinding = int(d->winding);
1782 }
1783 if (!qFuzzyCompare(d->depthBias, cbD->currentDepthBiasValues.first)
1784 || !qFuzzyCompare(d->slopeScaledDepthBias, cbD->currentDepthBiasValues.second))
1785 {
1786 [cbD->d->currentRenderPassEncoder setDepthBias: d->depthBias
1787 slopeScale: d->slopeScaledDepthBias
1788 clamp: 0.0f];
1789 cbD->currentDepthBiasValues = { d->depthBias, d->slopeScaledDepthBias };
1790 }
1791
1792 if (cbD->pushConstantsNeedRebind && !d->tess.enabled) {
1793 const int vsIdx = mtlPushConstantBufferIndex(d->vs);
1794 const int fsIdx = mtlPushConstantBufferIndex(d->fs);
1795 if (vsIdx >= 0 || fsIdx >= 0) {
1796 const quint32 blockSize = qMax(mtlPushConstantBlockSize(d->vs), mtlPushConstantBlockSize(d->fs));
1797 if (quint32(cbD->pushConstantData.size()) < blockSize)
1798 cbD->pushConstantData.resize(int(blockSize), 0);
1799 const NSUInteger total = NSUInteger(cbD->pushConstantData.size());
1800 if (vsIdx >= 0)
1801 [cbD->d->currentRenderPassEncoder setVertexBytes: cbD->pushConstantData.constData() length: total atIndex: NSUInteger(vsIdx)];
1802 if (fsIdx >= 0)
1803 [cbD->d->currentRenderPassEncoder setFragmentBytes: cbD->pushConstantData.constData() length: total atIndex: NSUInteger(fsIdx)];
1804 cbD->pushConstantsNeedRebind = false;
1805 }
1806 }
1807}
1808
1809void QRhiMetal::setGraphicsPipeline(QRhiCommandBuffer *cb, QRhiGraphicsPipeline *ps)
1810{
1811 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
1814
1816 return;
1817
1819 cbD->currentComputePipeline = nullptr;
1821
1822 if (cbD->hasCustomScissorSet && !psD->m_flags.testFlag(QRhiGraphicsPipeline::UsesScissor))
1824
1825 if (!psD->d->tess.enabled && !psD->d->tess.failed)
1827
1828 // mark work buffers that can now be safely reused as reusable
1829 // NOTE: These are usually empty unless tessellation or mutiview is used.
1830 for (QMetalBuffer *workBuf : psD->d->extraBufMgr.deviceLocalWorkBuffers) {
1831 if (workBuf && workBuf->lastActiveFrameSlot == currentFrameSlot)
1832 workBuf->lastActiveFrameSlot = -1;
1833 }
1834 for (QMetalBuffer *workBuf : psD->d->extraBufMgr.hostVisibleWorkBuffers) {
1835 if (workBuf && workBuf->lastActiveFrameSlot == currentFrameSlot)
1836 workBuf->lastActiveFrameSlot = -1;
1837 }
1838
1839 psD->lastActiveFrameSlot = currentFrameSlot;
1840}
1841
1842void QRhiMetal::setShaderResources(QRhiCommandBuffer *cb, QRhiShaderResourceBindings *srb,
1843 int dynamicOffsetCount,
1844 const QRhiCommandBuffer::DynamicOffset *dynamicOffsets)
1845{
1846 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
1850
1851 if (!srb) {
1852 if (gfxPsD)
1853 srb = gfxPsD->m_shaderResourceBindings;
1854 else
1855 srb = compPsD->m_shaderResourceBindings;
1856 }
1857
1859 bool hasSlottedResourceInSrb = false;
1860 bool hasDynamicOffsetInSrb = false;
1861 bool resNeedsRebind = false;
1862
1863 bool pipelineChanged = false;
1864 if (gfxPsD) {
1865 pipelineChanged = srbD->lastUsedGraphicsPipeline != gfxPsD;
1866 srbD->lastUsedGraphicsPipeline = gfxPsD;
1867 } else {
1868 pipelineChanged = srbD->lastUsedComputePipeline != compPsD;
1869 srbD->lastUsedComputePipeline = compPsD;
1870 }
1871
1872 // SPIRV-Cross buffer size buffers
1873 // Need to determine storage buffer sizes here as this is the last opportunity for storage
1874 // buffer bindings (offset, size) to be specified before draw / dispatch call
1875 const bool needsBufferSizeBuffer = (compPsD && compPsD->d->bufferSizeBuffer) || (gfxPsD && gfxPsD->d->bufferSizeBuffer);
1876 QMap<QRhiShaderResourceBinding::StageFlag, QMap<int, quint32>> storageBufferSizes;
1877
1878 // do buffer writes, figure out if we need to rebind, and mark as in-use
1879 for (int i = 0, ie = srbD->sortedBindings.count(); i != ie; ++i) {
1880 const QRhiShaderResourceBinding::Data *b = shaderResourceBindingData(srbD->sortedBindings.at(i));
1881 QMetalShaderResourceBindings::BoundResourceData &bd(srbD->boundResourceData[i]);
1882 switch (b->type) {
1883 case QRhiShaderResourceBinding::UniformBuffer:
1884 {
1885 QMetalBuffer *bufD = QRHI_RES(QMetalBuffer, b->u.ubuf.buf);
1886 Q_ASSERT(bufD->m_usage.testFlag(QRhiBuffer::UniformBuffer));
1887 sanityCheckResourceOwnership(bufD);
1889 if (bufD->d->slotted)
1890 hasSlottedResourceInSrb = true;
1891 if (b->u.ubuf.hasDynamicOffset)
1892 hasDynamicOffsetInSrb = true;
1893 if (bufD->generation != bd.ubuf.generation || bufD->m_id != bd.ubuf.id) {
1894 resNeedsRebind = true;
1895 bd.ubuf.id = bufD->m_id;
1896 bd.ubuf.generation = bufD->generation;
1897 }
1898 bufD->lastActiveFrameSlot = currentFrameSlot;
1899 }
1900 break;
1901 case QRhiShaderResourceBinding::SampledTexture:
1902 case QRhiShaderResourceBinding::Texture:
1903 case QRhiShaderResourceBinding::Sampler:
1904 {
1905 const QRhiShaderResourceBinding::Data::TextureAndOrSamplerData *data = &b->stex;
1906 if (bd.stex.d.size() != data->count()) {
1907 bd.stex.d.resize(data->count());
1908 resNeedsRebind = true;
1909 }
1910 for (int elem = 0; elem < data->count(); ++elem) {
1911 QMetalTexture *texD = QRHI_RES(QMetalTexture, data->texSamplers[elem].tex);
1912 QMetalSampler *samplerD = QRHI_RES(QMetalSampler, data->texSamplers[elem].sampler);
1913 Q_ASSERT(texD || samplerD);
1914 sanityCheckResourceOwnership(texD);
1915 sanityCheckResourceOwnership(samplerD);
1916 const quint64 texId = texD ? texD->m_id : 0;
1917 const uint texGen = texD ? texD->generation : 0;
1918 const quint64 samplerId = samplerD ? samplerD->m_id : 0;
1919 const uint samplerGen = samplerD ? samplerD->generation : 0;
1920 if (texGen != bd.stex.d[elem].texGeneration
1921 || texId != bd.stex.d[elem].texId
1922 || samplerGen != bd.stex.d[elem].samplerGeneration
1923 || samplerId != bd.stex.d[elem].samplerId)
1924 {
1925 resNeedsRebind = true;
1926 bd.stex.d[elem].texId = texId;
1927 bd.stex.d[elem].texGeneration = texGen;
1928 bd.stex.d[elem].samplerId = samplerId;
1929 bd.stex.d[elem].samplerGeneration = samplerGen;
1930 }
1931 if (texD)
1932 texD->lastActiveFrameSlot = currentFrameSlot;
1933 if (samplerD)
1934 samplerD->lastActiveFrameSlot = currentFrameSlot;
1935 }
1936 }
1937 break;
1938 case QRhiShaderResourceBinding::ImageLoad:
1939 case QRhiShaderResourceBinding::ImageStore:
1940 case QRhiShaderResourceBinding::ImageLoadStore:
1941 {
1942 QMetalTexture *texD = QRHI_RES(QMetalTexture, b->u.simage.tex);
1943 sanityCheckResourceOwnership(texD);
1944 if (texD->generation != bd.simage.generation || texD->m_id != bd.simage.id) {
1945 resNeedsRebind = true;
1946 bd.simage.id = texD->m_id;
1947 bd.simage.generation = texD->generation;
1948 }
1949 texD->lastActiveFrameSlot = currentFrameSlot;
1950 }
1951 break;
1952 case QRhiShaderResourceBinding::BufferLoad:
1953 case QRhiShaderResourceBinding::BufferStore:
1954 case QRhiShaderResourceBinding::BufferLoadStore:
1955 {
1956 QMetalBuffer *bufD = QRHI_RES(QMetalBuffer, b->u.sbuf.buf);
1957 Q_ASSERT(bufD->m_usage.testFlag(QRhiBuffer::StorageBuffer));
1958 sanityCheckResourceOwnership(bufD);
1959
1960 if (needsBufferSizeBuffer) {
1961 for (int i = 0; i < 6; ++i) {
1962 const QRhiShaderResourceBinding::StageFlag stage =
1963 QRhiShaderResourceBinding::StageFlag(1 << i);
1964 if (b->stage.testFlag(stage)) {
1965 storageBufferSizes[stage][b->binding] = b->u.sbuf.maybeSize ? b->u.sbuf.maybeSize : bufD->size();
1966 }
1967 }
1968 }
1969
1971 if (bufD->generation != bd.sbuf.generation || bufD->m_id != bd.sbuf.id) {
1972 resNeedsRebind = true;
1973 bd.sbuf.id = bufD->m_id;
1974 bd.sbuf.generation = bufD->generation;
1975 }
1976 bufD->lastActiveFrameSlot = currentFrameSlot;
1977 }
1978 break;
1979 default:
1980 Q_UNREACHABLE();
1981 break;
1982 }
1983 }
1984
1985 if (needsBufferSizeBuffer) {
1986 QMetalBuffer *bufD = nullptr;
1987 QVarLengthArray<std::pair<QMetalShader *, QRhiShaderResourceBinding::StageFlag>, 4> shaders;
1988
1989 if (compPsD) {
1990 bufD = compPsD->d->bufferSizeBuffer;
1991 Q_ASSERT(compPsD->d->cs.nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding));
1992 shaders.append({&compPsD->d->cs, QRhiShaderResourceBinding::StageFlag::ComputeStage});
1993 } else {
1994 bufD = gfxPsD->d->bufferSizeBuffer;
1995 if (gfxPsD->d->tess.enabled) {
1996
1997 // Assumptions
1998 // * We only use one of the compute vertex shader variants in a pipeline at any one time
1999 // * The vertex shader variants all have the same storage block bindings
2000 // * The vertex shader variants all have the same native resource binding map
2001 // * The vertex shader variants all have the same MslBufferSizeBufferBinding requirement
2002 // * The vertex shader variants all have the same MslBufferSizeBufferBinding binding
2003 // => We only need to use one vertex shader variant to generate the identical shader
2004 // resource bindings
2005 Q_ASSERT(gfxPsD->d->tess.compVs[0].desc.storageBlocks() == gfxPsD->d->tess.compVs[1].desc.storageBlocks());
2006 Q_ASSERT(gfxPsD->d->tess.compVs[0].desc.storageBlocks() == gfxPsD->d->tess.compVs[2].desc.storageBlocks());
2007 Q_ASSERT(gfxPsD->d->tess.compVs[0].nativeResourceBindingMap == gfxPsD->d->tess.compVs[1].nativeResourceBindingMap);
2008 Q_ASSERT(gfxPsD->d->tess.compVs[0].nativeResourceBindingMap == gfxPsD->d->tess.compVs[2].nativeResourceBindingMap);
2009 Q_ASSERT(gfxPsD->d->tess.compVs[0].nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding)
2010 == gfxPsD->d->tess.compVs[1].nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding));
2011 Q_ASSERT(gfxPsD->d->tess.compVs[0].nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding)
2012 == gfxPsD->d->tess.compVs[2].nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding));
2013 Q_ASSERT(gfxPsD->d->tess.compVs[0].nativeShaderInfo.extraBufferBindings[QShaderPrivate::MslBufferSizeBufferBinding]
2014 == gfxPsD->d->tess.compVs[1].nativeShaderInfo.extraBufferBindings[QShaderPrivate::MslBufferSizeBufferBinding]);
2015 Q_ASSERT(gfxPsD->d->tess.compVs[0].nativeShaderInfo.extraBufferBindings[QShaderPrivate::MslBufferSizeBufferBinding]
2016 == gfxPsD->d->tess.compVs[2].nativeShaderInfo.extraBufferBindings[QShaderPrivate::MslBufferSizeBufferBinding]);
2017
2018 if (gfxPsD->d->tess.compVs[0].nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding))
2019 shaders.append({&gfxPsD->d->tess.compVs[0], QRhiShaderResourceBinding::StageFlag::VertexStage});
2020
2021 if (gfxPsD->d->tess.compTesc.nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding))
2022 shaders.append({&gfxPsD->d->tess.compTesc, QRhiShaderResourceBinding::StageFlag::TessellationControlStage});
2023
2024 if (gfxPsD->d->tess.vertTese.nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding))
2025 shaders.append({&gfxPsD->d->tess.vertTese, QRhiShaderResourceBinding::StageFlag::TessellationEvaluationStage});
2026
2027 } else {
2028 if (gfxPsD->d->vs.nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding))
2029 shaders.append({&gfxPsD->d->vs, QRhiShaderResourceBinding::StageFlag::VertexStage});
2030 }
2031 if (gfxPsD->d->fs.nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding))
2032 shaders.append({&gfxPsD->d->fs, QRhiShaderResourceBinding::StageFlag::FragmentStage});
2033 }
2034
2035 quint32 offset = 0;
2036 for (const auto &shader : shaders) {
2037
2038 const int binding = shader.first->nativeShaderInfo.extraBufferBindings[QShaderPrivate::MslBufferSizeBufferBinding];
2039
2040 // if we don't have a srb entry for the buffer size buffer
2041 if (!(storageBufferSizes.contains(shader.second) && storageBufferSizes[shader.second].contains(binding))) {
2042
2043 int maxNativeBinding = 0;
2044 for (const QShaderDescription::StorageBlock &block : shader.first->desc.storageBlocks())
2045 maxNativeBinding = qMax(maxNativeBinding, shader.first->nativeResourceBindingMap[block.binding].first);
2046
2047 const int size = (maxNativeBinding + 1) * sizeof(int);
2048
2049 Q_ASSERT(offset + size <= bufD->size());
2050 srbD->sortedBindings.append(QRhiShaderResourceBinding::bufferLoad(binding, shader.second, bufD, offset, size));
2051
2052 QMetalShaderResourceBindings::BoundResourceData bd;
2053 bd.sbuf.id = bufD->m_id;
2054 bd.sbuf.generation = bufD->generation;
2055 srbD->boundResourceData.append(bd);
2056 }
2057
2058 // create the buffer size buffer data
2059 QVarLengthArray<int, 8> bufferSizeBufferData;
2060 Q_ASSERT(storageBufferSizes.contains(shader.second));
2061 const QMap<int, quint32> &sizes(storageBufferSizes[shader.second]);
2062 for (const QShaderDescription::StorageBlock &block : shader.first->desc.storageBlocks()) {
2063 const int index = shader.first->nativeResourceBindingMap[block.binding].first;
2064
2065 // if the native binding is -1, the buffer is present but not accessed in the shader
2066 if (index < 0)
2067 continue;
2068
2069 if (bufferSizeBufferData.size() <= index)
2070 bufferSizeBufferData.resize(index + 1);
2071
2072 Q_ASSERT(sizes.contains(block.binding));
2073 bufferSizeBufferData[index] = sizes[block.binding];
2074 }
2075
2076 QRhiBufferData data;
2077 const quint32 size = bufferSizeBufferData.size() * sizeof(int);
2078 data.assign(reinterpret_cast<const char *>(bufferSizeBufferData.constData()), size);
2079 Q_ASSERT(offset + size <= bufD->size());
2080 bufD->d->pendingUpdates[bufD->d->slotted ? currentFrameSlot : 0].append({ offset, data });
2081
2082 // buffer offsets must be 32byte aligned
2083 offset += ((size + 31) / 32) * 32;
2084 }
2085
2087 bufD->lastActiveFrameSlot = currentFrameSlot;
2088 }
2089
2090 // make sure the resources for the correct slot get bound
2091 const int resSlot = hasSlottedResourceInSrb ? currentFrameSlot : 0;
2092 if (hasSlottedResourceInSrb && cbD->currentResSlot != resSlot)
2093 resNeedsRebind = true;
2094
2095 const bool srbChanged = gfxPsD ? (cbD->currentGraphicsSrb != srbD) : (cbD->currentComputeSrb != srbD);
2096 const bool srbRebuilt = cbD->currentSrbGeneration != srbD->generation;
2097
2098 // dynamic uniform buffer offsets always trigger a rebind
2099 if (hasDynamicOffsetInSrb || resNeedsRebind || srbChanged || srbRebuilt || pipelineChanged) {
2100 const QShader::NativeResourceBindingMap *resBindMaps[SUPPORTED_STAGES] = { nullptr, nullptr, nullptr, nullptr, nullptr };
2101 const QMetalShader *shaders[SUPPORTED_STAGES] = { nullptr, nullptr, nullptr, nullptr, nullptr };
2102 if (gfxPsD) {
2103 cbD->currentGraphicsSrb = srbD;
2104 cbD->currentComputeSrb = nullptr;
2105 if (gfxPsD->d->tess.enabled) {
2106 // If tessellating, we don't know which compVs shader to use until the draw call is
2107 // made. They should all have the same native resource binding map, so pick one.
2108 Q_ASSERT(gfxPsD->d->tess.compVs[0].nativeResourceBindingMap == gfxPsD->d->tess.compVs[1].nativeResourceBindingMap);
2109 Q_ASSERT(gfxPsD->d->tess.compVs[0].nativeResourceBindingMap == gfxPsD->d->tess.compVs[2].nativeResourceBindingMap);
2110 resBindMaps[QMetalShaderResourceBindingsData::VERTEX] = &gfxPsD->d->tess.compVs[0].nativeResourceBindingMap;
2111 resBindMaps[QMetalShaderResourceBindingsData::TESSCTRL] = &gfxPsD->d->tess.compTesc.nativeResourceBindingMap;
2112 resBindMaps[QMetalShaderResourceBindingsData::TESSEVAL] = &gfxPsD->d->tess.vertTese.nativeResourceBindingMap;
2113 } else {
2114 resBindMaps[QMetalShaderResourceBindingsData::VERTEX] = &gfxPsD->d->vs.nativeResourceBindingMap;
2115 shaders[QMetalShaderResourceBindingsData::VERTEX] = &gfxPsD->d->vs;
2116 }
2117 resBindMaps[QMetalShaderResourceBindingsData::FRAGMENT] = &gfxPsD->d->fs.nativeResourceBindingMap;
2118 shaders[QMetalShaderResourceBindingsData::FRAGMENT] = &gfxPsD->d->fs;
2119 } else {
2120 cbD->currentGraphicsSrb = nullptr;
2121 cbD->currentComputeSrb = srbD;
2122 resBindMaps[QMetalShaderResourceBindingsData::COMPUTE] = &compPsD->d->cs.nativeResourceBindingMap;
2123 shaders[QMetalShaderResourceBindingsData::COMPUTE] = &compPsD->d->cs;
2124 }
2126 cbD->currentResSlot = resSlot;
2127
2128 const bool offsetOnlyChange = hasDynamicOffsetInSrb && !resNeedsRebind
2129 && !srbChanged && !srbRebuilt && !pipelineChanged;
2130 enqueueShaderResourceBindings(srbD, cbD, dynamicOffsetCount, dynamicOffsets, offsetOnlyChange,
2131 resBindMaps, shaders);
2132 }
2133}
2134
2135void QRhiMetal::setVertexInput(QRhiCommandBuffer *cb,
2136 int startBinding, int bindingCount, const QRhiCommandBuffer::VertexInput *bindings,
2137 QRhiBuffer *indexBuf, quint32 indexOffset, QRhiCommandBuffer::IndexFormat indexFormat)
2138{
2139 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
2141
2142 QRhiBatchedBindings<id<MTLBuffer> > buffers;
2143 QRhiBatchedBindings<NSUInteger> offsets;
2144 for (int i = 0; i < bindingCount; ++i) {
2145 QMetalBuffer *bufD = QRHI_RES(QMetalBuffer, bindings[i].first);
2147 bufD->lastActiveFrameSlot = currentFrameSlot;
2148 id<MTLBuffer> mtlbuf = bufD->d->buf[bufD->d->slotted ? currentFrameSlot : 0];
2149 buffers.feed(startBinding + i, mtlbuf);
2150 offsets.feed(startBinding + i, bindings[i].second);
2151 }
2152 buffers.finish();
2153 offsets.finish();
2154
2155 // same binding space for vertex and constant buffers - work it around
2157 // There's nothing guaranteeing setShaderResources() was called before
2158 // setVertexInput()... but whatever srb will get bound will have to be
2159 // layout-compatible anyways so maxBinding is the same.
2160 if (!srbD)
2161 srbD = QRHI_RES(QMetalShaderResourceBindings, cbD->currentGraphicsPipeline->shaderResourceBindings());
2162 const int firstVertexBinding = srbD->maxBinding + 1;
2163
2164 if (firstVertexBinding != cbD->d->currentFirstVertexBinding
2165 || buffers != cbD->d->currentVertexInputsBuffers
2166 || offsets != cbD->d->currentVertexInputOffsets)
2167 {
2168 cbD->d->currentFirstVertexBinding = firstVertexBinding;
2169 cbD->d->currentVertexInputsBuffers = buffers;
2170 cbD->d->currentVertexInputOffsets = offsets;
2171
2172 for (int i = 0, ie = buffers.batches.count(); i != ie; ++i) {
2173 const auto &bufferBatch(buffers.batches[i]);
2174 const auto &offsetBatch(offsets.batches[i]);
2175 [cbD->d->currentRenderPassEncoder setVertexBuffers:
2176 bufferBatch.resources.constData()
2177 offsets: offsetBatch.resources.constData()
2178 withRange: NSMakeRange(uint(firstVertexBinding) + bufferBatch.startBinding, NSUInteger(bufferBatch.resources.count()))];
2179 }
2180 }
2181
2182 if (indexBuf) {
2183 QMetalBuffer *ibufD = QRHI_RES(QMetalBuffer, indexBuf);
2185 ibufD->lastActiveFrameSlot = currentFrameSlot;
2186 cbD->currentIndexBuffer = ibufD;
2187 cbD->currentIndexOffset = indexOffset;
2188 cbD->currentIndexFormat = indexFormat;
2189 } else {
2190 cbD->currentIndexBuffer = nullptr;
2191 }
2192}
2193
2194// The size of the coordinate space viewports and scissors are specified in.
2195// Normally this is the pixel size of the render target. When a rasterization
2196// rate map is attached, Metal interprets viewports and scissors in the logical
2197// (screen space) coordinate system of the map, which is typically larger than
2198// the physical texture size. This is what foveated rendering on visionOS uses.
2199QSize QRhiMetal::outputSizeForTarget(QRhiRenderTarget *target)
2200{
2201 QRhiShadingRateMap *srm = nullptr;
2202 switch (target->resourceType()) {
2203 case QRhiResource::TextureRenderTarget:
2204 srm = QRHI_RES(QMetalTextureRenderTarget, target)->m_desc.shadingRateMap();
2205 break;
2206 case QRhiResource::SwapChainRenderTarget:
2207 srm = QRHI_RES(QMetalSwapChainRenderTarget, target)->swapChain()->shadingRateMap();
2208 break;
2209 default:
2210 break;
2211 }
2212
2213 if (srm) {
2214 const QSize logicalSize = srm->logicalSize();
2215 if (logicalSize.isValid())
2216 return logicalSize;
2217 }
2218
2219 return target->pixelSize();
2220}
2221
2223{
2224 cbD->hasCustomScissorSet = false;
2225
2226 const QSize outputSize = outputSizeForTarget(cbD->currentTarget);
2227 std::array<float, 4> vp = cbD->currentViewport.viewport();
2228 float x = 0, y = 0, w = 0, h = 0;
2229
2230 if (qFuzzyIsNull(vp[2]) && qFuzzyIsNull(vp[3])) {
2231 x = 0;
2232 y = 0;
2233 w = outputSize.width();
2234 h = outputSize.height();
2235 } else {
2236 // x,y is top-left in MTLScissorRect but bottom-left in QRhiScissor
2237 qrhi_toTopLeftRenderTargetRect<Bounded>(outputSize, vp, &x, &y, &w, &h);
2238 }
2239
2240 MTLScissorRect s;
2241 s.x = NSUInteger(x);
2242 s.y = NSUInteger(y);
2243 s.width = NSUInteger(w);
2244 s.height = NSUInteger(h);
2245 [cbD->d->currentRenderPassEncoder setScissorRect: s];
2246}
2247
2248void QRhiMetal::setViewport(QRhiCommandBuffer *cb, const QRhiViewport &viewport)
2249{
2250 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
2252 const QSize outputSize = outputSizeForTarget(cbD->currentTarget);
2253
2254 // x,y is top-left in MTLViewportRect but bottom-left in QRhiViewport
2255 float x, y, w, h;
2256 if (!qrhi_toTopLeftRenderTargetRect<UnBounded>(outputSize, viewport.viewport(), &x, &y, &w, &h))
2257 return;
2258
2259 MTLViewport vp;
2260 vp.originX = double(x);
2261 vp.originY = double(y);
2262 vp.width = double(w);
2263 vp.height = double(h);
2264 vp.znear = double(viewport.minDepth());
2265 vp.zfar = double(viewport.maxDepth());
2266
2267 [cbD->d->currentRenderPassEncoder setViewport: vp];
2268
2269 cbD->currentViewport = viewport;
2271 && !cbD->currentGraphicsPipeline->m_flags.testFlag(QRhiGraphicsPipeline::UsesScissor))
2272 {
2274 }
2275}
2276
2277void QRhiMetal::setScissor(QRhiCommandBuffer *cb, const QRhiScissor &scissor)
2278{
2279 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
2281 Q_ASSERT(!cbD->currentGraphicsPipeline
2282 || cbD->currentGraphicsPipeline->m_flags.testFlag(QRhiGraphicsPipeline::UsesScissor));
2283 const QSize outputSize = outputSizeForTarget(cbD->currentTarget);
2284
2285 // x,y is top-left in MTLScissorRect but bottom-left in QRhiScissor
2286 int x, y, w, h;
2287 if (!qrhi_toTopLeftRenderTargetRect<Bounded>(outputSize, scissor.scissor(), &x, &y, &w, &h))
2288 return;
2289
2290 MTLScissorRect s;
2291 s.x = NSUInteger(x);
2292 s.y = NSUInteger(y);
2293 s.width = NSUInteger(w);
2294 s.height = NSUInteger(h);
2295
2296 [cbD->d->currentRenderPassEncoder setScissorRect: s];
2297
2298 cbD->hasCustomScissorSet = true;
2299 cbD->currentScissor = scissor;
2300}
2301
2302void QRhiMetal::setBlendConstants(QRhiCommandBuffer *cb, const QColor &c)
2303{
2304 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
2306
2307 [cbD->d->currentRenderPassEncoder setBlendColorRed: c.redF()
2308 green: c.greenF() blue: c.blueF() alpha: c.alphaF()];
2309
2310 cbD->hasBlendConstantsSet = true;
2311 cbD->currentBlendConstants = c;
2312}
2313
2314void QRhiMetal::setStencilRef(QRhiCommandBuffer *cb, quint32 refValue)
2315{
2316 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
2318
2319 [cbD->d->currentRenderPassEncoder setStencilReferenceValue: refValue];
2320
2321 cbD->hasStencilRefSet = true;
2322 cbD->currentStencilRef = refValue;
2323}
2324
2325void QRhiMetal::setPushConstants(QRhiCommandBuffer *cb, quint32 offset, quint32 size, const void *data)
2326{
2327 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
2329
2330 // setVertexBytes() and friends take the whole block every time, so a shadow
2331 // copy is needed to honour an update of part of it.
2332 auto patch = [cbD, offset, size, data](quint32 blockSize) {
2333 const quint32 total = qMax(blockSize, offset + size);
2334 if (quint32(cbD->pushConstantData.size()) < total)
2335 cbD->pushConstantData.resize(int(total), 0);
2336 memcpy(cbD->pushConstantData.data() + offset, data, size);
2337 return total;
2338 };
2339
2342 if (!psD)
2343 return;
2344 const int idx = mtlPushConstantBufferIndex(psD->d->cs);
2345 if (idx < 0) {
2346 qWarning("No pipeline with a push constant block is active; setPushConstants ignored");
2347 return;
2348 }
2349 const quint32 total = patch(mtlPushConstantBlockSize(psD->d->cs));
2350 [cbD->d->currentComputePassEncoder setBytes: cbD->pushConstantData.constData() length: total atIndex: NSUInteger(idx)];
2351 } else {
2353 if (!psD)
2354 return;
2355 if (psD->d->tess.enabled) {
2356 qWarning("Push constants are not supported with tessellation on Metal");
2357 return;
2358 }
2359 const int vsIdx = mtlPushConstantBufferIndex(psD->d->vs);
2360 const int fsIdx = mtlPushConstantBufferIndex(psD->d->fs);
2361 if (vsIdx < 0 && fsIdx < 0) {
2362 qWarning("No pipeline with a push constant block is active; setPushConstants ignored");
2363 return;
2364 }
2365 const quint32 total = patch(qMax(mtlPushConstantBlockSize(psD->d->vs), mtlPushConstantBlockSize(psD->d->fs)));
2366 if (vsIdx >= 0)
2367 [cbD->d->currentRenderPassEncoder setVertexBytes: cbD->pushConstantData.constData() length: total atIndex: NSUInteger(vsIdx)];
2368 if (fsIdx >= 0)
2369 [cbD->d->currentRenderPassEncoder setFragmentBytes: cbD->pushConstantData.constData() length: total atIndex: NSUInteger(fsIdx)];
2370 }
2371}
2372
2373void QRhiMetal::setShadingRate(QRhiCommandBuffer *cb, const QSize &coarsePixelSize)
2374{
2375 Q_UNUSED(cb);
2376 Q_UNUSED(coarsePixelSize);
2377}
2378
2380{
2381 switch (cbD->currentTarget->resourceType()) {
2382 case QRhiResource::SwapChainRenderTarget:
2383 return QRHI_RES(QMetalSwapChainRenderTarget, cbD->currentTarget)->d;
2384 case QRhiResource::TextureRenderTarget:
2385 return QRHI_RES(QMetalTextureRenderTarget, cbD->currentTarget)->d;
2386 default:
2387 return nullptr;
2388 }
2389}
2390
2391// Memoryless attachments only exist for the duration of a single encoder, there
2392// is nowhere to store them to.
2393static inline bool canStoreAttachment(id<MTLTexture> tex)
2394{
2395 return tex && tex.storageMode != MTLStorageModeMemoryless;
2396}
2397
2398// When the pass continues on a new encoder afterwards, the contents have to be
2399// stored, not just resolved or discarded. An attachment with a resolve texture
2400// can only use the resolving store actions.
2401static inline MTLStoreAction interruptionStoreAction(MTLStoreAction finalAction, bool passIsEnding)
2402{
2403 if (passIsEnding)
2404 return finalAction;
2405 return finalAction == MTLStoreActionDontCare ? MTLStoreActionStore
2406 : MTLStoreActionStoreAndMultisampleResolve;
2407}
2408
2409// Store actions are deferred (MTLStoreActionUnknown) for attachments whose
2410// contents may need to outlive the encoder, so they have to be finalized before
2411// the encoder ends.
2412// Store actions cannot be mutated otherwise, and for a resolve attachment the
2413// only legal choices are the two resolving ones.
2415{
2416 for (const auto &[index, finalAction] : std::as_const(cbD->d->deferredColorStoreActions))
2417 [cbD->d->currentRenderPassEncoder setColorStoreAction:
2418 interruptionStoreAction(finalAction, passIsEnding) atIndex: index];
2419
2420 if (cbD->d->deferredDepthStoreAction != MTLStoreActionUnknown) {
2421 [cbD->d->currentRenderPassEncoder setDepthStoreAction:
2422 interruptionStoreAction(cbD->d->deferredDepthStoreAction, passIsEnding)];
2423 }
2424 if (cbD->d->deferredStencilStoreAction != MTLStoreActionUnknown) {
2425 [cbD->d->currentRenderPassEncoder setStencilStoreAction:
2426 interruptionStoreAction(cbD->d->deferredStencilStoreAction, passIsEnding)];
2427 }
2428}
2429
2430// Ends the render command encoder such that the pass can be continued on a new
2431// encoder afterwards.
2433{
2435 for (qsizetype i = 0; i < cbD->d->openDebugGroups.size(); ++i)
2436 [cbD->d->currentRenderPassEncoder popDebugGroup];
2437 [cbD->d->currentRenderPassEncoder endEncoding];
2438 cbD->d->currentRenderPassEncoder = nil;
2439}
2440
2443 id<MTLComputeCommandEncoder> maybeComputeEncoder)
2444{
2445 if (cbD->d->currentRenderPassEncoder)
2447
2448 if (!maybeComputeEncoder)
2449 maybeComputeEncoder = [cbD->d->cb computeCommandEncoder];
2450
2451 return maybeComputeEncoder;
2452}
2453
2455 id<MTLComputeCommandEncoder> computeEncoder)
2456{
2457 if (computeEncoder) {
2458 [computeEncoder endEncoding];
2459 computeEncoder = nil;
2460 }
2461
2463 Q_ASSERT(rtD);
2464
2465 // A memoryless attachment cannot be loaded, its contents are lost (see
2466 // NoTransientBacking), even when the store action is not DontCare, which is
2467 // the case with a resolve.
2468 const auto canLoad = [](MTLRenderPassAttachmentDescriptor *att) {
2469 return att.storeAction != MTLStoreActionDontCare && canStoreAttachment(att.texture);
2470 };
2471
2472 QVarLengthArray<MTLLoadAction, 4> oldColorLoad;
2473 for (uint i = 0; i < uint(rtD->colorAttCount); ++i) {
2474 oldColorLoad.append(cbD->d->currentPassRpDesc.colorAttachments[i].loadAction);
2475 if (canLoad(cbD->d->currentPassRpDesc.colorAttachments[i]))
2476 cbD->d->currentPassRpDesc.colorAttachments[i].loadAction = MTLLoadActionLoad;
2477 }
2478
2479 MTLLoadAction oldDepthLoad;
2480 MTLLoadAction oldStencilLoad;
2481 if (rtD->dsAttCount) {
2482 oldDepthLoad = cbD->d->currentPassRpDesc.depthAttachment.loadAction;
2483 if (canLoad(cbD->d->currentPassRpDesc.depthAttachment))
2484 cbD->d->currentPassRpDesc.depthAttachment.loadAction = MTLLoadActionLoad;
2485
2486 oldStencilLoad = cbD->d->currentPassRpDesc.stencilAttachment.loadAction;
2487 if (canLoad(cbD->d->currentPassRpDesc.stencilAttachment))
2488 cbD->d->currentPassRpDesc.stencilAttachment.loadAction = MTLLoadActionLoad;
2489 }
2490
2491 // The state below is not tied to the pipeline, so it is not restored by the
2492 // callers when they reactivate it on the new encoder. Preserve it here,
2493 // otherwise the pass would silently continue with a full-target viewport,
2494 // no scissor, and default blend constants and stencil reference.
2495 const QRhiViewport prevViewport = cbD->currentViewport;
2496 const bool prevHasScissor = cbD->hasCustomScissorSet;
2497 const QRhiScissor prevScissor = cbD->currentScissor;
2498 const bool prevHasBlendConstants = cbD->hasBlendConstantsSet;
2499 const QColor prevBlendConstants = cbD->currentBlendConstants;
2500 const bool prevHasStencilRef = cbD->hasStencilRefSet;
2501 const quint32 prevStencilRef = cbD->currentStencilRef;
2502 // Whether the pipeline was relying on the viewport-derived default scissor,
2503 // which setViewport() below cannot reapply on its own: it only does so when
2504 // a pipeline is already current, and there is none at that point.
2505 const bool prevHasDefaultScissor = cbD->currentGraphicsPipeline
2506 && !cbD->currentGraphicsPipeline->flags().testFlag(QRhiGraphicsPipeline::UsesScissor);
2507 // Rebound when the callers reactivate the pipeline, or when the next
2508 // pipeline with a push constant block is set.
2509 const QVarLengthArray<char, 128> prevPushConstantData = cbD->pushConstantData;
2510
2511 cbD->d->currentRenderPassEncoder = [cbD->d->cb renderCommandEncoderWithDescriptor: cbD->d->currentPassRpDesc];
2513
2514 for (const QByteArray &name : std::as_const(cbD->d->openDebugGroups))
2515 [cbD->d->currentRenderPassEncoder pushDebugGroup: [NSString stringWithUTF8String: name.constData()]];
2516
2517 cbD->pushConstantData = prevPushConstantData;
2518 cbD->pushConstantsNeedRebind = !prevPushConstantData.isEmpty();
2519
2520 // Must come before the callers reactivate the pipeline: setScissor()
2521 // expects no pipeline to be current yet, and setDefaultScissor() consults
2522 // the viewport.
2523 if (!qFuzzyIsNull(prevViewport.viewport()[2]) || !qFuzzyIsNull(prevViewport.viewport()[3]))
2524 rhiD->setViewport(cbD, prevViewport);
2525 if (prevHasScissor)
2526 rhiD->setScissor(cbD, prevScissor);
2527 else if (prevHasDefaultScissor)
2528 rhiD->setDefaultScissor(cbD);
2529 if (prevHasBlendConstants)
2530 rhiD->setBlendConstants(cbD, prevBlendConstants);
2531 if (prevHasStencilRef)
2532 rhiD->setStencilRef(cbD, prevStencilRef);
2533
2534 for (uint i = 0; i < uint(rtD->colorAttCount); ++i) {
2535 cbD->d->currentPassRpDesc.colorAttachments[i].loadAction = oldColorLoad[i];
2536 }
2537
2538 if (rtD->dsAttCount) {
2539 cbD->d->currentPassRpDesc.depthAttachment.loadAction = oldDepthLoad;
2540 cbD->d->currentPassRpDesc.stencilAttachment.loadAction = oldStencilLoad;
2541 }
2542
2543}
2544
2546{
2547 QMetalCommandBuffer *cbD = args.cbD;
2549 if (graphicsPipeline->d->tess.failed)
2550 return;
2551
2552 const bool indexed = args.type != TessDrawArgs::NonIndexed;
2553 const quint32 instanceCount = indexed ? args.drawIndexed.instanceCount : args.draw.instanceCount;
2554 const quint32 vertexOrIndexCount = indexed ? args.drawIndexed.indexCount : args.draw.vertexCount;
2555
2556 QMetalGraphicsPipelineData::Tessellation &tess(graphicsPipeline->d->tess);
2557 QMetalGraphicsPipelineData::ExtraBufferManager &extraBufMgr(graphicsPipeline->d->extraBufMgr);
2558 const quint32 patchCount = tess.patchCountForDrawCall(vertexOrIndexCount, instanceCount);
2559 QMetalBuffer *vertOutBuf = nullptr;
2560 QMetalBuffer *tescOutBuf = nullptr;
2561 QMetalBuffer *tescPatchOutBuf = nullptr;
2562 QMetalBuffer *tescFactorBuf = nullptr;
2563 QMetalBuffer *tescParamsBuf = nullptr;
2564 id<MTLComputeCommandEncoder> vertTescComputeEncoder
2565 = tempComputeEncoder(this, cbD, cbD->d->tessellationComputeEncoder);
2566 cbD->d->tessellationComputeEncoder = vertTescComputeEncoder;
2567
2568 // Step 1: vertex shader (as compute)
2569 {
2570 id<MTLComputeCommandEncoder> computeEncoder = vertTescComputeEncoder;
2571 QShader::Variant shaderVariant = QShader::NonIndexedVertexAsComputeShader;
2572 if (args.type == TessDrawArgs::U16Indexed)
2573 shaderVariant = QShader::UInt16IndexedVertexAsComputeShader;
2574 else if (args.type == TessDrawArgs::U32Indexed)
2575 shaderVariant = QShader::UInt32IndexedVertexAsComputeShader;
2576 const int varIndex = QMetalGraphicsPipelineData::Tessellation::vsCompVariantToIndex(shaderVariant);
2577 id<MTLComputePipelineState> computePipelineState = tess.vsCompPipeline(this, shaderVariant);
2578 [computeEncoder setComputePipelineState: computePipelineState];
2579
2580 // Make uniform buffers, textures, and samplers (meant for the
2581 // vertex stage from the client's point of view) visible in the
2582 // "vertex as compute" shader
2583 cbD->d->currentComputePassEncoder = computeEncoder;
2585 cbD->d->currentComputePassEncoder = nil;
2586
2587 const QMap<int, int> &ebb(tess.compVs[varIndex].nativeShaderInfo.extraBufferBindings);
2588 const int outputBufferBinding = ebb.value(QShaderPrivate::MslTessVertTescOutputBufferBinding, -1);
2589 const int indexBufferBinding = ebb.value(QShaderPrivate::MslTessVertIndicesBufferBinding, -1);
2590
2591 if (outputBufferBinding >= 0) {
2592 const quint32 workBufSize = tess.vsCompOutputBufferSize(vertexOrIndexCount, instanceCount);
2593 vertOutBuf = extraBufMgr.acquireWorkBuffer(this, workBufSize);
2594 if (!vertOutBuf)
2595 return;
2596 [computeEncoder setBuffer: vertOutBuf->d->buf[0] offset: 0 atIndex: outputBufferBinding];
2597 }
2598
2599 if (indexBufferBinding >= 0)
2600 [computeEncoder setBuffer: (id<MTLBuffer>) args.drawIndexed.indexBuffer offset: 0 atIndex: indexBufferBinding];
2601
2602 for (int i = 0, ie = cbD->d->currentVertexInputsBuffers.batches.count(); i != ie; ++i) {
2603 const auto &bufferBatch(cbD->d->currentVertexInputsBuffers.batches[i]);
2604 const auto &offsetBatch(cbD->d->currentVertexInputOffsets.batches[i]);
2605 [computeEncoder setBuffers: bufferBatch.resources.constData()
2606 offsets: offsetBatch.resources.constData()
2607 withRange: NSMakeRange(uint(cbD->d->currentFirstVertexBinding) + bufferBatch.startBinding, NSUInteger(bufferBatch.resources.count()))];
2608 }
2609
2610 if (indexed) {
2611 [computeEncoder setStageInRegion: MTLRegionMake2D(args.drawIndexed.vertexOffset, args.drawIndexed.firstInstance,
2612 args.drawIndexed.indexCount, args.drawIndexed.instanceCount)];
2613 } else {
2614 [computeEncoder setStageInRegion: MTLRegionMake2D(args.draw.firstVertex, args.draw.firstInstance,
2615 args.draw.vertexCount, args.draw.instanceCount)];
2616 }
2617
2618 [computeEncoder dispatchThreads: MTLSizeMake(vertexOrIndexCount, instanceCount, 1)
2619 threadsPerThreadgroup: MTLSizeMake(computePipelineState.threadExecutionWidth, 1, 1)];
2620 }
2621
2622 // Step 2: tessellation control shader (as compute)
2623 {
2624 id<MTLComputeCommandEncoder> computeEncoder = vertTescComputeEncoder;
2625 id<MTLComputePipelineState> computePipelineState = tess.tescCompPipeline(this);
2626 [computeEncoder setComputePipelineState: computePipelineState];
2627
2628 cbD->d->currentComputePassEncoder = computeEncoder;
2630 cbD->d->currentComputePassEncoder = nil;
2631
2632 const QMap<int, int> &ebb(tess.compTesc.nativeShaderInfo.extraBufferBindings);
2633 const int outputBufferBinding = ebb.value(QShaderPrivate::MslTessVertTescOutputBufferBinding, -1);
2634 const int patchOutputBufferBinding = ebb.value(QShaderPrivate::MslTessTescPatchOutputBufferBinding, -1);
2635 const int tessFactorBufferBinding = ebb.value(QShaderPrivate::MslTessTescTessLevelBufferBinding, -1);
2636 const int paramsBufferBinding = ebb.value(QShaderPrivate::MslTessTescParamsBufferBinding, -1);
2637 const int inputBufferBinding = ebb.value(QShaderPrivate::MslTessTescInputBufferBinding, -1);
2638
2639 if (outputBufferBinding >= 0) {
2640 const quint32 workBufSize = tess.tescCompOutputBufferSize(patchCount);
2641 tescOutBuf = extraBufMgr.acquireWorkBuffer(this, workBufSize);
2642 if (!tescOutBuf)
2643 return;
2644 [computeEncoder setBuffer: tescOutBuf->d->buf[0] offset: 0 atIndex: outputBufferBinding];
2645 }
2646
2647 if (patchOutputBufferBinding >= 0) {
2648 const quint32 workBufSize = tess.tescCompPatchOutputBufferSize(patchCount);
2649 tescPatchOutBuf = extraBufMgr.acquireWorkBuffer(this, workBufSize);
2650 if (!tescPatchOutBuf)
2651 return;
2652 [computeEncoder setBuffer: tescPatchOutBuf->d->buf[0] offset: 0 atIndex: patchOutputBufferBinding];
2653 }
2654
2655 if (tessFactorBufferBinding >= 0) {
2656 tescFactorBuf = extraBufMgr.acquireWorkBuffer(this, patchCount * sizeof(MTLQuadTessellationFactorsHalf));
2657 [computeEncoder setBuffer: tescFactorBuf->d->buf[0] offset: 0 atIndex: tessFactorBufferBinding];
2658 }
2659
2660 if (paramsBufferBinding >= 0) {
2661 struct {
2662 quint32 inControlPointCount;
2663 quint32 patchCount;
2664 } params;
2665 tescParamsBuf = extraBufMgr.acquireWorkBuffer(this, sizeof(params), QMetalGraphicsPipelineData::ExtraBufferManager::WorkBufType::HostVisible);
2666 if (!tescParamsBuf)
2667 return;
2668 params.inControlPointCount = tess.inControlPointCount;
2669 params.patchCount = patchCount;
2670 id<MTLBuffer> paramsBuf = tescParamsBuf->d->buf[0];
2671 char *p = reinterpret_cast<char *>([paramsBuf contents]);
2672 memcpy(p, &params, sizeof(params));
2673 [computeEncoder setBuffer: paramsBuf offset: 0 atIndex: paramsBufferBinding];
2674 }
2675
2676 if (vertOutBuf && inputBufferBinding >= 0)
2677 [computeEncoder setBuffer: vertOutBuf->d->buf[0] offset: 0 atIndex: inputBufferBinding];
2678
2679 int sgSize = int(computePipelineState.threadExecutionWidth);
2680 int wgSize = std::lcm(tess.outControlPointCount, sgSize);
2681 while (wgSize > caps.maxThreadGroupSize) {
2682 sgSize /= 2;
2683 wgSize = std::lcm(tess.outControlPointCount, sgSize);
2684 }
2685 [computeEncoder dispatchThreads: MTLSizeMake(patchCount * tess.outControlPointCount, 1, 1)
2686 threadsPerThreadgroup: MTLSizeMake(wgSize, 1, 1)];
2687 }
2688
2689 // Much of the state in the QMetalCommandBuffer is going to be reset
2690 // when we get a new render encoder. Save what we need. (cheaper than
2691 // starting to walk over the srb again)
2692 const QMetalShaderResourceBindingsData resourceBindings = cbD->d->currentShaderResourceBindingState;
2693
2694 endTempComputeEncoding(this, cbD, cbD->d->tessellationComputeEncoder);
2695 cbD->d->tessellationComputeEncoder = nil;
2696
2697 // Step 3: tessellation evaluation (as vertex) + fragment shader
2698 {
2699 // No need to call tess.teseFragRenderPipeline because it was done
2700 // once and we know the result is stored in the standard place
2701 // (graphicsPipeline->d->ps).
2702
2704 id<MTLRenderCommandEncoder> renderEncoder = cbD->d->currentRenderPassEncoder;
2705
2708
2709 const QMap<int, int> &ebb(tess.compTesc.nativeShaderInfo.extraBufferBindings);
2710 const int outputBufferBinding = ebb.value(QShaderPrivate::MslTessVertTescOutputBufferBinding, -1);
2711 const int patchOutputBufferBinding = ebb.value(QShaderPrivate::MslTessTescPatchOutputBufferBinding, -1);
2712 const int tessFactorBufferBinding = ebb.value(QShaderPrivate::MslTessTescTessLevelBufferBinding, -1);
2713
2714 if (outputBufferBinding >= 0 && tescOutBuf)
2715 [renderEncoder setVertexBuffer: tescOutBuf->d->buf[0] offset: 0 atIndex: outputBufferBinding];
2716
2717 if (patchOutputBufferBinding >= 0 && tescPatchOutBuf)
2718 [renderEncoder setVertexBuffer: tescPatchOutBuf->d->buf[0] offset: 0 atIndex: patchOutputBufferBinding];
2719
2720 if (tessFactorBufferBinding >= 0 && tescFactorBuf) {
2721 [renderEncoder setTessellationFactorBuffer: tescFactorBuf->d->buf[0] offset: 0 instanceStride: 0];
2722 [renderEncoder setVertexBuffer: tescFactorBuf->d->buf[0] offset: 0 atIndex: tessFactorBufferBinding];
2723 }
2724
2725 [cbD->d->currentRenderPassEncoder drawPatches: tess.outControlPointCount
2726 patchStart: 0
2727 patchCount: patchCount
2728 patchIndexBuffer: nil
2729 patchIndexBufferOffset: 0
2730 instanceCount: 1
2731 baseInstance: 0];
2732 }
2733}
2734
2735void QRhiMetal::adjustForMultiViewDraw(quint32 *instanceCount, QRhiCommandBuffer *cb)
2736{
2737 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
2738 const int multiViewCount = cbD->currentGraphicsPipeline->m_multiViewCount;
2739 if (multiViewCount <= 1)
2740 return;
2741
2742 const QMap<int, int> &ebb(cbD->currentGraphicsPipeline->d->vs.nativeShaderInfo.extraBufferBindings);
2743 const int viewMaskBufBinding = ebb.value(QShaderPrivate::MslMultiViewMaskBufferBinding, -1);
2744 if (viewMaskBufBinding == -1) {
2745 qWarning("No extra buffer for multiview in the vertex shader; was it built with --view-count specified?");
2746 return;
2747 }
2748 struct {
2749 quint32 viewOffset;
2750 quint32 viewCount;
2751 } multiViewInfo;
2752 multiViewInfo.viewOffset = 0;
2753 multiViewInfo.viewCount = quint32(multiViewCount);
2754 QMetalBuffer *buf = cbD->currentGraphicsPipeline->d->extraBufMgr.acquireWorkBuffer(this, sizeof(multiViewInfo),
2756 if (buf) {
2757 id<MTLBuffer> mtlbuf = buf->d->buf[0];
2758 char *p = reinterpret_cast<char *>([mtlbuf contents]);
2759 memcpy(p, &multiViewInfo, sizeof(multiViewInfo));
2760 [cbD->d->currentRenderPassEncoder setVertexBuffer: mtlbuf offset: 0 atIndex: viewMaskBufBinding];
2761 // The instance count is adjusted for layered rendering. The vertex shader is expected to contain something like:
2762 // uint gl_ViewIndex = spvViewMask[0] + (gl_InstanceIndex - gl_BaseInstance) % spvViewMask[1];
2763 // where spvViewMask is the buffer with multiViewInfo passed in above.
2764 *instanceCount *= multiViewCount;
2765 }
2766}
2767
2768void QRhiMetal::draw(QRhiCommandBuffer *cb, quint32 vertexCount,
2769 quint32 instanceCount, quint32 firstVertex, quint32 firstInstance)
2770{
2771 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
2773
2774 if (cbD->currentGraphicsPipeline->d->tess.enabled) {
2775 TessDrawArgs a;
2776 a.cbD = cbD;
2777 a.type = TessDrawArgs::NonIndexed;
2778 a.draw.vertexCount = vertexCount;
2779 a.draw.instanceCount = instanceCount;
2780 a.draw.firstVertex = firstVertex;
2781 a.draw.firstInstance = firstInstance;
2783 return;
2784 }
2785
2786 adjustForMultiViewDraw(&instanceCount, cb);
2787
2788 if (caps.baseVertexAndInstance) {
2789 [cbD->d->currentRenderPassEncoder drawPrimitives: cbD->currentGraphicsPipeline->d->primitiveType
2790 vertexStart: firstVertex vertexCount: vertexCount instanceCount: instanceCount baseInstance: firstInstance];
2791 } else {
2792 [cbD->d->currentRenderPassEncoder drawPrimitives: cbD->currentGraphicsPipeline->d->primitiveType
2793 vertexStart: firstVertex vertexCount: vertexCount instanceCount: instanceCount];
2794 }
2795}
2796
2797void QRhiMetal::drawIndexed(QRhiCommandBuffer *cb, quint32 indexCount,
2798 quint32 instanceCount, quint32 firstIndex, qint32 vertexOffset, quint32 firstInstance)
2799{
2800 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
2802
2803 if (!cbD->currentIndexBuffer)
2804 return;
2805
2806 const quint32 indexOffset = cbD->currentIndexOffset + firstIndex * (cbD->currentIndexFormat == QRhiCommandBuffer::IndexUInt16 ? 2 : 4);
2807 Q_ASSERT(indexOffset == aligned(indexOffset, 4u));
2808
2810 id<MTLBuffer> mtlibuf = ibufD->d->buf[ibufD->d->slotted ? currentFrameSlot : 0];
2811
2812 if (cbD->currentGraphicsPipeline->d->tess.enabled) {
2813 TessDrawArgs a;
2814 a.cbD = cbD;
2815 a.type = cbD->currentIndexFormat == QRhiCommandBuffer::IndexUInt16 ? TessDrawArgs::U16Indexed : TessDrawArgs::U32Indexed;
2816 a.drawIndexed.indexCount = indexCount;
2817 a.drawIndexed.instanceCount = instanceCount;
2818 a.drawIndexed.firstIndex = firstIndex;
2819 a.drawIndexed.vertexOffset = vertexOffset;
2820 a.drawIndexed.firstInstance = firstInstance;
2821 a.drawIndexed.indexBuffer = mtlibuf;
2823 return;
2824 }
2825
2826 adjustForMultiViewDraw(&instanceCount, cb);
2827
2828 if (caps.baseVertexAndInstance) {
2829 [cbD->d->currentRenderPassEncoder drawIndexedPrimitives: cbD->currentGraphicsPipeline->d->primitiveType
2830 indexCount: indexCount
2831 indexType: cbD->currentIndexFormat == QRhiCommandBuffer::IndexUInt16 ? MTLIndexTypeUInt16 : MTLIndexTypeUInt32
2832 indexBuffer: mtlibuf
2833 indexBufferOffset: indexOffset
2834 instanceCount: instanceCount
2835 baseVertex: vertexOffset
2836 baseInstance: firstInstance];
2837 } else {
2838 [cbD->d->currentRenderPassEncoder drawIndexedPrimitives: cbD->currentGraphicsPipeline->d->primitiveType
2839 indexCount: indexCount
2840 indexType: cbD->currentIndexFormat == QRhiCommandBuffer::IndexUInt16 ? MTLIndexTypeUInt16 : MTLIndexTypeUInt32
2841 indexBuffer: mtlibuf
2842 indexBufferOffset: indexOffset
2843 instanceCount: instanceCount];
2844 }
2845}
2846
2847// Returns null when the ICB path is usable, otherwise the reason why not.
2849{
2850 if (!caps.indirectCommandBuffers)
2851 return "indirect command buffers are not supported on this device";
2853 || !cbD->currentGraphicsPipeline->m_flags.testFlag(QRhiGraphicsPipeline::UsesIndirectDraws))
2854 {
2855 return "the current graphics pipeline was not created with UsesIndirectDraws";
2856 }
2857 if (cbD->currentGraphicsPipeline->d->tess.enabled)
2858 return "the current graphics pipeline uses tessellation";
2860 return "the shaders of the current graphics pipeline sample textures but have no "
2861 "argument buffer variant, which Metal requires for a pipeline that supports "
2862 "indirect command buffers; rebuild them with qsb --msl-argument-buffers "
2863 "(or use MSLARGUMENTBUFFERS with qt_add_shaders)";
2864 }
2865 return nullptr;
2866}
2867
2869{
2871 return false;
2872
2873 if (!d->icbEncodePipeline) {
2874 NSError *err = nil;
2875 NSString *src = [NSString stringWithUTF8String:s_icbEncodeMsl];
2876 MTLCompileOptions *opts = [MTLCompileOptions new];
2877 opts.languageVersion = MTLLanguageVersion2_1;
2878 id<MTLLibrary> lib = [d->dev newLibraryWithSource:src options:opts error:&err];
2879 [opts release];
2880 if (!lib) {
2881 qWarning("Failed to compile ICB encode kernel: %s",
2882 qPrintable(QString::fromNSString(err.localizedDescription)));
2883 d->icbSetupFailed = true;
2884 return false;
2885 }
2886 d->icbEncodeFunction = [lib newFunctionWithName:@"encode_icb"];
2887 d->icbEncodeFunctionU32 = [lib newFunctionWithName:@"encode_icb_indexed_u32"];
2888 d->icbEncodeFunctionU16 = [lib newFunctionWithName:@"encode_icb_indexed_u16"];
2889 [lib release];
2890 if (!d->icbEncodeFunction || !d->icbEncodeFunctionU32 || !d->icbEncodeFunctionU16) {
2891 qWarning("ICB encode kernel functions not found");
2892 d->icbSetupFailed = true;
2893 return false;
2894 }
2895 NSError *errU32 = nil;
2896 NSError *errU16 = nil;
2897 d->icbEncodePipeline = [d->dev newComputePipelineStateWithFunction:d->icbEncodeFunction error:&err];
2898 d->icbEncodePipelineU32 = [d->dev newComputePipelineStateWithFunction:d->icbEncodeFunctionU32 error:&errU32];
2899 d->icbEncodePipelineU16 = [d->dev newComputePipelineStateWithFunction:d->icbEncodeFunctionU16 error:&errU16];
2900 if (!d->icbEncodePipeline || !d->icbEncodePipelineU32 || !d->icbEncodePipelineU16) {
2901 NSError *firstErr = !d->icbEncodePipeline ? err
2902 : (!d->icbEncodePipelineU32 ? errU32 : errU16);
2903 qWarning("Failed to create ICB encode compute pipeline: %s",
2904 qPrintable(QString::fromNSString(firstErr.localizedDescription)));
2905 d->icbSetupFailed = true;
2906 return false;
2907 }
2908 }
2909
2910 if (!d->icbRangeBuffer) {
2911 d->icbRangeBuffer = [d->dev newBufferWithLength:sizeof(MTLIndirectCommandBufferExecutionRange)
2912 options:MTLResourceStorageModePrivate];
2913 static constexpr quint32 noCount = 0xFFFFFFFFu;
2914 d->icbNoCountBuffer = [d->dev newBufferWithBytes:&noCount
2915 length:sizeof(noCount)
2916 options:MTLResourceStorageModeShared];
2917 if (!d->icbRangeBuffer || !d->icbNoCountBuffer) {
2918 qWarning("Failed to create ICB helper buffers");
2919 d->icbSetupFailed = true;
2920 return false;
2921 }
2922 }
2923
2924 return true;
2925}
2926
2927bool QRhiMetal::prepareIcb(quint32 maxDrawCount)
2928{
2930 return false;
2931
2932 if (!d->icb || d->icbCapacity < maxDrawCount) {
2933 if (d->icb) {
2936 e.lastActiveFrameSlot = currentFrameSlot;
2937 e.stagingIcbBuffer.icb = d->icb;
2938 e.stagingIcbBuffer.argBuffer = d->icbArgumentBuffer;
2939 d->releaseQueue.append(e);
2940 }
2941 d->icb = nil;
2942 d->icbArgumentBuffer = nil;
2943
2944 MTLIndirectCommandBufferDescriptor *icbDesc = [MTLIndirectCommandBufferDescriptor new];
2945 icbDesc.commandTypes = MTLIndirectCommandTypeDraw | MTLIndirectCommandTypeDrawIndexed;
2946 icbDesc.inheritPipelineState = YES;
2947 icbDesc.inheritBuffers = YES;
2948 icbDesc.maxVertexBufferBindCount = 0;
2949 icbDesc.maxFragmentBufferBindCount = 0;
2950 d->icb = [d->dev newIndirectCommandBufferWithDescriptor:icbDesc
2951 maxCommandCount:maxDrawCount
2952 options:MTLResourceStorageModePrivate];
2953 [icbDesc release];
2954 if (!d->icb) {
2955 qWarning("Failed to create MTLIndirectCommandBuffer");
2956 d->icbCapacity = 0;
2957 return false;
2958 }
2959 d->icbCapacity = maxDrawCount;
2960
2961 id<MTLArgumentEncoder> argEnc = [d->icbEncodeFunction newArgumentEncoderWithBufferIndex:1];
2962 d->icbArgumentBuffer = [d->dev newBufferWithLength:argEnc.encodedLength
2963 options:MTLResourceStorageModeShared];
2964 [argEnc setArgumentBuffer:d->icbArgumentBuffer offset:0];
2965 [argEnc setIndirectCommandBuffer:d->icb atIndex:0];
2966 [argEnc release];
2967 }
2968
2969 return true;
2970}
2971
2972// Encodes maxDrawCount entries of indirectBufMtl into targetIcb on an already
2973// open compute encoder, writing the encoded count to targetRangeBuffer.
2974// countBufMtl is optional: when null all maxDrawCount commands are encoded.
2976 id<MTLComputeCommandEncoder> computeEncoder,
2977 id<MTLIndirectCommandBuffer> targetIcb,
2978 id<MTLBuffer> targetArgBuffer,
2979 id<MTLBuffer> targetRangeBuffer,
2980 bool indexed,
2981 QRhiCommandBuffer::IndexFormat indexFormat,
2982 MTLPrimitiveType primitiveType,
2983 id<MTLBuffer> indirectBufMtl, quint32 indirectBufferOffset,
2984 id<MTLBuffer> indexBufMtl, quint32 indexBufferOffset,
2985 id<MTLBuffer> countBufMtl, quint32 countBufferOffset,
2986 quint32 maxDrawCount, quint32 stride)
2987{
2988 id<MTLComputePipelineState> computePipeline = d->icbEncodePipeline;
2989 if (indexed) {
2990 computePipeline = indexFormat == QRhiCommandBuffer::IndexUInt16
2991 ? d->icbEncodePipelineU16 : d->icbEncodePipelineU32;
2992 }
2993 uint32_t maxDrawCountVal = maxDrawCount;
2994 uint32_t metalPrimType = uint32_t(primitiveType);
2995 uint32_t strideVal = stride;
2996
2997 [computeEncoder setComputePipelineState:computePipeline];
2998 [computeEncoder setBuffer:indirectBufMtl offset:indirectBufferOffset atIndex:0];
2999 [computeEncoder setBuffer:targetArgBuffer offset:0 atIndex:1];
3000 [computeEncoder setBytes:&maxDrawCountVal length:sizeof(uint32_t) atIndex:2];
3001 if (indexed)
3002 [computeEncoder setBuffer:indexBufMtl offset:indexBufferOffset atIndex:3];
3003 [computeEncoder setBytes:&metalPrimType length:sizeof(uint32_t) atIndex:4];
3004 [computeEncoder setBytes:&strideVal length:sizeof(uint32_t) atIndex:5];
3005 [computeEncoder setBuffer:countBufMtl ? countBufMtl : d->icbNoCountBuffer
3006 offset:countBufMtl ? countBufferOffset : 0
3007 atIndex:6];
3008 [computeEncoder setBuffer:targetRangeBuffer offset:0 atIndex:7];
3009 [computeEncoder useResource:targetIcb usage:MTLResourceUsageWrite];
3010 [computeEncoder useResource:indirectBufMtl usage:MTLResourceUsageRead];
3011 if (indexed)
3012 [computeEncoder useResource:indexBufMtl usage:MTLResourceUsageRead];
3013
3014 NSUInteger tw = computePipeline.threadExecutionWidth;
3015 [computeEncoder dispatchThreads:MTLSizeMake(maxDrawCount, 1, 1)
3016 threadsPerThreadgroup:MTLSizeMake(tw, 1, 1)];
3017}
3018
3019// Encodes up to maxDrawCount commands into the shared ICB with a compute
3020// dispatch, then executes them on the render encoder. countBufMtl is optional:
3021// when null, all maxDrawCount commands are executed, otherwise the number of
3022// draws is min(maxDrawCount, <uint32 at countBufferOffset>), computed on the
3023// GPU. Returns false when the ICB could not be set up, in which case nothing
3024// was recorded and the pass is left untouched.
3025bool QRhiMetal::icbDraw(QMetalCommandBuffer *cbD, bool indexed,
3026 QMetalBuffer *indirectBufD, quint32 indirectBufferOffset,
3027 QMetalBuffer *countBufD, quint32 countBufferOffset,
3028 quint32 maxDrawCount, quint32 stride)
3029{
3030 if (!maxDrawCount)
3031 return true;
3032
3033 QMetalBuffer *indexBufD = cbD->currentIndexBuffer;
3034 if (indexed && !indexBufD)
3035 return false;
3036
3037 if (!prepareIcb(maxDrawCount))
3038 return false;
3039
3041 indirectBufD->lastActiveFrameSlot = currentFrameSlot;
3042 id<MTLBuffer> indirectBufMtl = indirectBufD->d->buf[indirectBufD->d->slotted ? currentFrameSlot : 0];
3043
3044 id<MTLBuffer> countBufMtl = nil;
3045 if (countBufD) {
3047 countBufD->lastActiveFrameSlot = currentFrameSlot;
3048 countBufMtl = countBufD->d->buf[countBufD->d->slotted ? currentFrameSlot : 0];
3049 }
3050
3051 // Everything below survives the interruption only if saved and restored.
3053 const QMetalShaderResourceBindingsData savedResourceBindings = cbD->d->currentShaderResourceBindingState;
3054 const int savedFirstVertexBinding = cbD->d->currentFirstVertexBinding;
3055 const auto savedVertexBuffers = cbD->d->currentVertexInputsBuffers;
3056 const auto savedVertexOffsets = cbD->d->currentVertexInputOffsets;
3057 const quint32 savedIndexOffset = cbD->currentIndexOffset;
3058 const QRhiCommandBuffer::IndexFormat savedIndexFormat = cbD->currentIndexFormat;
3059 id<MTLBuffer> indexBufMtl = indexed
3060 ? indexBufD->d->buf[indexBufD->d->slotted ? currentFrameSlot : 0] : nil;
3061
3062 // End the current render encoder to make room for the compute pass.
3064
3065 id<MTLComputeCommandEncoder> computeEncoder = [cbD->d->cb computeCommandEncoder];
3066 encodeIcbWithCompute(d, computeEncoder, d->icb, d->icbArgumentBuffer, d->icbRangeBuffer,
3067 indexed, savedIndexFormat, savedPipeline->d->primitiveType,
3068 indirectBufMtl, indirectBufferOffset,
3069 indexBufMtl, savedIndexOffset,
3070 countBufMtl, countBufferOffset,
3071 maxDrawCount, stride);
3072
3073 // Restart the render pass with Load actions to preserve existing content.
3074 endTempComputeEncoding(this, cbD, computeEncoder);
3075
3076 // Restore pipeline, shader resources, and vertex bindings on the new encoder.
3079 QMetalShaderResourceBindingsData::VERTEX, &savedResourceBindings);
3081 QMetalShaderResourceBindingsData::FRAGMENT, &savedResourceBindings);
3082
3083 if (savedFirstVertexBinding >= 0) {
3084 cbD->d->currentFirstVertexBinding = savedFirstVertexBinding;
3085 cbD->d->currentVertexInputsBuffers = savedVertexBuffers;
3086 cbD->d->currentVertexInputOffsets = savedVertexOffsets;
3087 for (int i = 0, ie = savedVertexBuffers.batches.count(); i != ie; ++i) {
3088 const auto &bufferBatch(savedVertexBuffers.batches[i]);
3089 const auto &offsetBatch(savedVertexOffsets.batches[i]);
3090 [cbD->d->currentRenderPassEncoder setVertexBuffers:
3091 bufferBatch.resources.constData()
3092 offsets: offsetBatch.resources.constData()
3093 withRange: NSMakeRange(uint(savedFirstVertexBinding) + bufferBatch.startBinding,
3094 NSUInteger(bufferBatch.resources.count()))];
3095 }
3096 }
3097
3098 cbD->currentIndexBuffer = indexBufD;
3099 cbD->currentIndexOffset = savedIndexOffset;
3100 cbD->currentIndexFormat = savedIndexFormat;
3101
3102 // Declare buffer dependencies and execute the GPU-encoded ICB. The range to
3103 // execute is read from icbRangeBuffer, which the kernel just wrote.
3104 [cbD->d->currentRenderPassEncoder useResource:indirectBufMtl
3105 usage:MTLResourceUsageRead
3106 stages:MTLRenderStageVertex | MTLRenderStageFragment];
3107 if (indexed) {
3108 [cbD->d->currentRenderPassEncoder useResource:indexBufMtl
3109 usage:MTLResourceUsageRead
3110 stages:MTLRenderStageVertex | MTLRenderStageFragment];
3111 }
3112 [cbD->d->currentRenderPassEncoder executeCommandsInBuffer:d->icb
3113 indirectBuffer:d->icbRangeBuffer
3114 indirectBufferOffset:0];
3115 return true;
3116}
3117
3118// The ICB (Indirect Command Buffer) path encodes the draw commands on the GPU
3119// and executes them with a single executeCommandsInBuffer, which removes the
3120// per-draw CPU overhead. The encoding needs a compute pass, so the render pass
3121// has to be interrupted and restarted: that costs about 100-150 microseconds,
3122// whereas an individual indirect draw call takes 1-2. Hence only taking this
3123// path for large batches.
3124static constexpr quint32 ICB_DRAW_COUNT_THRESHOLD = 128;
3125
3126void QRhiMetal::drawIndirect(QRhiCommandBuffer *cb, QRhiBuffer *indirectBuffer,
3127 quint32 indirectBufferOffset, quint32 drawCount, quint32 stride)
3128{
3129 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
3131
3132 QMetalBuffer *indirectBufD = QRHI_RES(QMetalBuffer, indirectBuffer);
3134 indirectBufD->lastActiveFrameSlot = currentFrameSlot;
3135 id<MTLBuffer> indirectBufMtl = indirectBufD->d->buf[indirectBufD->d->slotted ? currentFrameSlot : 0];
3136
3137 if (drawCount > ICB_DRAW_COUNT_THRESHOLD && !icbUnavailableReason(cbD)
3138 && icbDraw(cbD, false, indirectBufD, indirectBufferOffset, nullptr, 0, drawCount, stride))
3139 {
3140 return;
3141 }
3142
3143 // CPU-side for-loop fallback: used when ICB is not applicable or setup failed.
3144 NSUInteger offset = indirectBufferOffset;
3145 for (quint32 i = 0; i < drawCount; ++i) {
3146 [cbD->d->currentRenderPassEncoder drawPrimitives: cbD->currentGraphicsPipeline->d->primitiveType
3147 indirectBuffer: indirectBufMtl
3148 indirectBufferOffset: offset];
3149 offset += stride;
3150 }
3151}
3152
3153void QRhiMetal::drawIndexedIndirect(QRhiCommandBuffer *cb, QRhiBuffer *indirectBuffer,
3154 quint32 indirectBufferOffset, quint32 drawCount, quint32 stride)
3155{
3156 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
3158
3159 if (!cbD->currentIndexBuffer)
3160 return;
3161
3162 QMetalBuffer *indexBufD = cbD->currentIndexBuffer;
3163 id<MTLBuffer> indexBufMtl = indexBufD->d->buf[indexBufD->d->slotted ? currentFrameSlot : 0];
3164
3165 QMetalBuffer *indirectBufD = QRHI_RES(QMetalBuffer, indirectBuffer);
3167 indirectBufD->lastActiveFrameSlot = currentFrameSlot;
3168 id<MTLBuffer> indirectBufMtl = indirectBufD->d->buf[indirectBufD->d->slotted ? currentFrameSlot : 0];
3169
3170 if (drawCount > ICB_DRAW_COUNT_THRESHOLD && !icbUnavailableReason(cbD)
3171 && icbDraw(cbD, true, indirectBufD, indirectBufferOffset, nullptr, 0, drawCount, stride))
3172 {
3173 return;
3174 }
3175
3176 // CPU-side for-loop fallback: used when ICB is not applicable or setup failed.
3177 NSUInteger offset = indirectBufferOffset;
3178 for (quint32 i = 0; i < drawCount; ++i) {
3179 [cbD->d->currentRenderPassEncoder drawIndexedPrimitives: cbD->currentGraphicsPipeline->d->primitiveType
3180 indexType: cbD->currentIndexFormat == QRhiCommandBuffer::IndexUInt16 ? MTLIndexTypeUInt16 : MTLIndexTypeUInt32
3181 indexBuffer: indexBufMtl
3182 indexBufferOffset: cbD->currentIndexOffset
3183 indirectBuffer: indirectBufMtl
3184 indirectBufferOffset: offset];
3185 offset += stride;
3186 }
3187}
3188
3189void QRhiMetal::debugMarkBegin(QRhiCommandBuffer *cb, const QByteArray &name)
3190{
3191 if (!debugMarkers)
3192 return;
3193
3194 NSString *str = [NSString stringWithUTF8String: name.constData()];
3195 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
3196 switch (cbD->recordingPass) {
3197 case QMetalCommandBuffer::RenderPass:
3198 [cbD->d->currentRenderPassEncoder pushDebugGroup: str];
3199 cbD->d->openDebugGroups.append(name);
3200 break;
3201 case QMetalCommandBuffer::ComputePass:
3202 [cbD->d->currentComputePassEncoder pushDebugGroup: str];
3203 break;
3204 default:
3205 [cbD->d->cb pushDebugGroup: str];
3206 break;
3207 }
3208}
3209
3210void QRhiMetal::debugMarkEnd(QRhiCommandBuffer *cb)
3211{
3212 if (!debugMarkers)
3213 return;
3214
3215 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
3216 switch (cbD->recordingPass) {
3217 case QMetalCommandBuffer::RenderPass:
3218 [cbD->d->currentRenderPassEncoder popDebugGroup];
3219 if (!cbD->d->openDebugGroups.isEmpty())
3220 cbD->d->openDebugGroups.removeLast();
3221 break;
3222 case QMetalCommandBuffer::ComputePass:
3223 [cbD->d->currentComputePassEncoder popDebugGroup];
3224 break;
3225 default:
3226 [cbD->d->cb popDebugGroup];
3227 break;
3228 }
3229}
3230
3231void QRhiMetal::debugMarkMsg(QRhiCommandBuffer *cb, const QByteArray &msg)
3232{
3233 if (!debugMarkers)
3234 return;
3235
3236 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
3237 NSString *str = [NSString stringWithUTF8String: msg.constData()];
3238 switch (cbD->recordingPass) {
3239 case QMetalCommandBuffer::RenderPass:
3240 [cbD->d->currentRenderPassEncoder insertDebugSignpost: str];
3241 break;
3242 case QMetalCommandBuffer::ComputePass:
3243 [cbD->d->currentComputePassEncoder insertDebugSignpost: str];
3244 break;
3245 default:
3246 break;
3247 }
3248}
3249
3250const QRhiNativeHandles *QRhiMetal::nativeHandles(QRhiCommandBuffer *cb)
3251{
3252 return QRHI_RES(QMetalCommandBuffer, cb)->nativeHandles();
3253}
3254
3255void QRhiMetal::beginExternal(QRhiCommandBuffer *cb)
3256{
3257 Q_UNUSED(cb);
3258}
3259
3260void QRhiMetal::endExternal(QRhiCommandBuffer *cb)
3261{
3262 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
3264}
3265
3266double QRhiMetal::lastCompletedGpuTime(QRhiCommandBuffer *cb)
3267{
3268 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
3269 return cbD->d->lastGpuTime;
3270}
3271
3272QRhi::FrameOpResult QRhiMetal::beginFrame(QRhiSwapChain *swapChain, QRhi::BeginFrameFlags flags)
3273{
3274 Q_UNUSED(flags);
3275
3276 QMetalSwapChain *swapChainD = QRHI_RES(QMetalSwapChain, swapChain);
3277 currentSwapChain = swapChainD;
3278 currentFrameSlot = swapChainD->currentFrameSlot;
3279
3280 // If we are too far ahead, block. This is also what ensures that any
3281 // resource used in the previous frame for this slot is now not in use
3282 // anymore by the GPU.
3283 dispatch_semaphore_wait(swapChainD->d->sem[currentFrameSlot], DISPATCH_TIME_FOREVER);
3284
3285 // Do this also for any other swapchain's commands with the same frame slot
3286 // While this reduces concurrency, it keeps resource usage safe: swapchain
3287 // A starting its frame 0, followed by swapchain B starting its own frame 0
3288 // will make B wait for A's frame 0 commands, so if a resource is written
3289 // in B's frame or when B checks for pending resource releases, that won't
3290 // mess up A's in-flight commands (as they are not in flight anymore).
3291 for (QMetalSwapChain *sc : std::as_const(swapchains)) {
3292 if (sc != swapChainD)
3293 sc->waitUntilCompleted(currentFrameSlot); // wait+signal
3294 }
3295
3296 [d->captureScope beginScope];
3297
3298 swapChainD->cbWrapper.d->cb = d->newCommandBuffer();
3299
3301 if (swapChainD->samples > 1) {
3302 colorAtt.tex = swapChainD->d->msaaTex[currentFrameSlot];
3303 colorAtt.needsDrawableForResolveTex = true;
3304 } else {
3305 colorAtt.needsDrawableForTex = true;
3306 }
3307
3308 swapChainD->rtWrapper.d->fb.colorAtt[0] = colorAtt;
3309 swapChainD->rtWrapper.d->fb.dsTex = swapChainD->ds ? swapChainD->ds->d->tex : nil;
3310 swapChainD->rtWrapper.d->fb.dsResolveTex = nil;
3311 swapChainD->rtWrapper.d->fb.hasStencil = swapChainD->ds ? true : false;
3312 swapChainD->rtWrapper.d->fb.depthNeedsStore = false;
3313
3314 if (swapChainD->ds)
3315 swapChainD->ds->lastActiveFrameSlot = currentFrameSlot;
3316
3317 d->argBufPool[currentFrameSlot].offset = 0;
3318 d->resetAndResizeBufferStagingArea(currentFrameSlot);
3319 d->globalFrameId += 1;
3320
3322 swapChainD->cbWrapper.resetState(swapChainD->d->lastGpuTime[currentFrameSlot]);
3323 swapChainD->d->lastGpuTime[currentFrameSlot] = 0;
3325
3326 return QRhi::FrameOpSuccess;
3327}
3328
3329QRhi::FrameOpResult QRhiMetal::endFrame(QRhiSwapChain *swapChain, QRhi::EndFrameFlags flags)
3330{
3331 QMetalSwapChain *swapChainD = QRHI_RES(QMetalSwapChain, swapChain);
3332 Q_ASSERT(currentSwapChain == swapChainD);
3333
3334 // Keep strong reference to command buffer
3335 id<MTLCommandBuffer> commandBuffer = swapChainD->cbWrapper.d->cb;
3336
3337 __block int thisFrameSlot = currentFrameSlot;
3338 [commandBuffer addCompletedHandler: ^(id<MTLCommandBuffer> cb) {
3339 swapChainD->d->lastGpuTime[thisFrameSlot] += cb.GPUEndTime - cb.GPUStartTime;
3340 dispatch_semaphore_signal(swapChainD->d->sem[thisFrameSlot]);
3341 }];
3342
3344 // When Metal API validation diagnostics is enabled in Xcode the texture is
3345 // released before the command buffer is done with it. Manually keep it alive
3346 // to work around this.
3347 id<MTLTexture> drawableTexture = [swapChainD->d->curDrawable.texture retain];
3348 [commandBuffer addCompletedHandler:^(id<MTLCommandBuffer>) {
3349 [drawableTexture release];
3350 }];
3351#endif
3352
3353 if (flags.testFlag(QRhi::SkipPresent)) {
3354 // Just need to commit, that's it
3355 [commandBuffer commit];
3356 } else {
3357 if (id<CAMetalDrawable> drawable = swapChainD->d->curDrawable) {
3358 // Got something to present
3359 if (swapChainD->d->layer.presentsWithTransaction) {
3360 [commandBuffer commit];
3361 // Keep strong reference to Metal layer
3362 auto *metalLayer = swapChainD->d->layer;
3363 auto presentWithTransaction = ^{
3364 [commandBuffer waitUntilScheduled];
3365 // If the layer has been resized while we waited to be scheduled we bail out,
3366 // as the drawable is no longer valid for the layer, and we'll get a follow-up
3367 // display with the right size. We know we are on the main thread here, which
3368 // means we can access the layer directly. We also know that the layer is valid,
3369 // since the block keeps a strong reference to it, compared to the QRhiSwapChain
3370 // that can go away under our feet by the time we're scheduled.
3371 const auto surfaceSize = QSizeF::fromCGSize(metalLayer.bounds.size) * metalLayer.contentsScale;
3372 const auto textureSize = QSizeF(drawable.texture.width, drawable.texture.height);
3373 if (textureSize == surfaceSize) {
3374 [drawable present];
3375 } else {
3376 qCDebug(QRHI_LOG_INFO) << "Skipping" << drawable << "due to texture size"
3377 << textureSize << "not matching surface size" << surfaceSize;
3378 }
3379 };
3380
3381 if (NSThread.currentThread == NSThread.mainThread) {
3382 presentWithTransaction();
3383 } else {
3384 auto *qtMetalLayer = qt_objc_cast<QMetalLayer*>(swapChainD->d->layer);
3385 Q_ASSERT(qtMetalLayer);
3386 // Let the main thread present the drawable from displayLayer
3387 qtMetalLayer.mainThreadPresentation = presentWithTransaction;
3388 }
3389 } else {
3390 // Keep strong reference to Metal layer so it's valid in the block
3391 auto *qtMetalLayer = qt_objc_cast<QMetalLayer*>(swapChainD->d->layer);
3392 [commandBuffer addScheduledHandler:^(id<MTLCommandBuffer>) {
3393 if (qtMetalLayer) {
3394 // The schedule handler comes in on the com.Metal.CompletionQueueDispatch
3395 // thread, which means we might be racing against a display cycle on the
3396 // main thread. If the displayLayer is already in progress, we don't want
3397 // to step on its toes.
3398 if (qtMetalLayer.displayLock.tryLockForRead()) {
3399 [drawable present];
3400 qtMetalLayer.displayLock.unlock();
3401 } else {
3402 qCDebug(QRHI_LOG_INFO) << "Skipping" << drawable
3403 << "due to" << qtMetalLayer << "needing display";
3404 }
3405 } else {
3406 [drawable present];
3407 }
3408 }];
3409 [commandBuffer commit];
3410 }
3411 } else {
3412 // Still need to commit, even if we don't have a drawable
3413 [commandBuffer commit];
3414 }
3415
3416 swapChainD->currentFrameSlot = (swapChainD->currentFrameSlot + 1) % QMTL_FRAMES_IN_FLIGHT;
3417 }
3418
3419 // Must not hold on to the drawable, regardless of needsPresent
3420 [swapChainD->d->curDrawable release];
3421 swapChainD->d->curDrawable = nil;
3422
3423 [d->captureScope endScope];
3424
3425 swapChainD->frameCount += 1;
3426 currentSwapChain = nullptr;
3427 return QRhi::FrameOpSuccess;
3428}
3429
3430QRhi::FrameOpResult QRhiMetal::beginOffscreenFrame(QRhiCommandBuffer **cb, QRhi::BeginFrameFlags flags)
3431{
3432 Q_UNUSED(flags);
3433
3434 currentFrameSlot = (currentFrameSlot + 1) % QMTL_FRAMES_IN_FLIGHT;
3435
3436 for (QMetalSwapChain *sc : std::as_const(swapchains))
3437 sc->waitUntilCompleted(currentFrameSlot);
3438
3439 d->ofr.active = true;
3440 *cb = &d->ofr.cbWrapper;
3441 d->ofr.cbWrapper.d->cb = d->newCommandBuffer();
3442
3443 d->argBufPool[currentFrameSlot].offset = 0;
3444 d->resetAndResizeBufferStagingArea(currentFrameSlot);
3445 d->globalFrameId += 1;
3446
3448 d->ofr.cbWrapper.resetState(d->ofr.lastGpuTime);
3449 d->ofr.lastGpuTime = 0;
3451
3452 return QRhi::FrameOpSuccess;
3453}
3454
3455QRhi::FrameOpResult QRhiMetal::endOffscreenFrame(QRhi::EndFrameFlags flags)
3456{
3457 Q_UNUSED(flags);
3458 Q_ASSERT(d->ofr.active);
3459 d->ofr.active = false;
3460
3461 id<MTLCommandBuffer> cb = d->ofr.cbWrapper.d->cb;
3462 [cb commit];
3463
3464 // offscreen frames wait for completion, unlike swapchain ones
3465 [cb waitUntilCompleted];
3466
3467 d->ofr.lastGpuTime += cb.GPUEndTime - cb.GPUStartTime;
3468
3470
3471 return QRhi::FrameOpSuccess;
3472}
3473
3475{
3476 id<MTLCommandBuffer> cb = nil;
3477 QMetalSwapChain *swapChainD = nullptr;
3478 if (inFrame) {
3479 if (d->ofr.active) {
3480 Q_ASSERT(!currentSwapChain);
3481 Q_ASSERT(d->ofr.cbWrapper.recordingPass == QMetalCommandBuffer::NoPass);
3482 cb = d->ofr.cbWrapper.d->cb;
3483 } else {
3484 Q_ASSERT(currentSwapChain);
3485 swapChainD = currentSwapChain;
3486 Q_ASSERT(swapChainD->cbWrapper.recordingPass == QMetalCommandBuffer::NoPass);
3487 cb = swapChainD->cbWrapper.d->cb;
3488 }
3489 }
3490
3491 for (QMetalSwapChain *sc : std::as_const(swapchains)) {
3492 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
3493 if (currentSwapChain && sc == currentSwapChain && i == currentFrameSlot) {
3494 // no wait as this is the thing we're going to be commit below and
3495 // beginFrame decremented sem already and going to be signaled by endFrame
3496 continue;
3497 }
3498 sc->waitUntilCompleted(i);
3499 }
3500 }
3501
3502 if (cb) {
3503 [cb commit];
3504 [cb waitUntilCompleted];
3505 }
3506
3507 if (inFrame) {
3508 if (d->ofr.active) {
3509 d->ofr.lastGpuTime += cb.GPUEndTime - cb.GPUStartTime;
3510 d->ofr.cbWrapper.d->cb = d->newCommandBuffer();
3511 } else {
3512 swapChainD->d->lastGpuTime[currentFrameSlot] += cb.GPUEndTime - cb.GPUStartTime;
3513 swapChainD->cbWrapper.d->cb = d->newCommandBuffer();
3514 }
3515 }
3516
3518
3520
3521 return QRhi::FrameOpSuccess;
3522}
3523
3525 const QColor &colorClearValue,
3526 const QRhiDepthStencilClearValue &depthStencilClearValue,
3527 int colorAttCount,
3528 QRhiShadingRateMap *shadingRateMap)
3529{
3530 MTLRenderPassDescriptor *rp = [MTLRenderPassDescriptor renderPassDescriptor];
3531 MTLClearColor c = MTLClearColorMake(colorClearValue.redF(), colorClearValue.greenF(), colorClearValue.blueF(),
3532 colorClearValue.alphaF());
3533
3534 for (uint i = 0; i < uint(colorAttCount); ++i) {
3535 rp.colorAttachments[i].loadAction = MTLLoadActionClear;
3536 rp.colorAttachments[i].storeAction = MTLStoreActionStore;
3537 rp.colorAttachments[i].clearColor = c;
3538 }
3539
3540 if (hasDepthStencil) {
3541 rp.depthAttachment.loadAction = MTLLoadActionClear;
3542 rp.depthAttachment.storeAction = MTLStoreActionDontCare;
3543 rp.stencilAttachment.loadAction = MTLLoadActionClear;
3544 rp.stencilAttachment.storeAction = MTLStoreActionDontCare;
3545 rp.depthAttachment.clearDepth = double(depthStencilClearValue.depthClearValue());
3546 rp.stencilAttachment.clearStencil = depthStencilClearValue.stencilClearValue();
3547 }
3548
3549 if (shadingRateMap)
3550 rp.rasterizationRateMap = QRHI_RES(QMetalShadingRateMap, shadingRateMap)->d->rateMap;
3551
3552 return rp;
3553}
3554
3555qsizetype QRhiMetal::subresUploadByteSize(const QRhiTextureSubresourceUploadDescription &subresDesc) const
3556{
3557 qsizetype size = 0;
3558 const qsizetype imageSizeBytes = subresDesc.image().isNull() ?
3559 subresDesc.data().size() : subresDesc.image().sizeInBytes();
3560 if (imageSizeBytes > 0)
3561 size += aligned<qsizetype>(imageSizeBytes, QRhiMetalData::TEXBUF_ALIGN);
3562 return size;
3563}
3564
3565void QRhiMetal::enqueueSubresUpload(QMetalTexture *texD, void *mp, void *blitEncPtr,
3566 int layer, int level, const QRhiTextureSubresourceUploadDescription &subresDesc,
3567 qsizetype *curOfs)
3568{
3569 const QPoint dp = subresDesc.destinationTopLeft();
3570 const QByteArray rawData = subresDesc.data();
3571 QImage img = subresDesc.image();
3572 const bool is3D = texD->m_flags.testFlag(QRhiTexture::ThreeDimensional);
3573 id<MTLBlitCommandEncoder> blitEnc = (id<MTLBlitCommandEncoder>) blitEncPtr;
3574
3575 if (!img.isNull()) {
3576 const qsizetype fullImageSizeBytes = img.sizeInBytes();
3577 QSize size = img.size();
3578 int bpl = img.bytesPerLine();
3579
3580 if (!subresDesc.sourceSize().isEmpty() || !subresDesc.sourceTopLeft().isNull()) {
3581 const int sx = subresDesc.sourceTopLeft().x();
3582 const int sy = subresDesc.sourceTopLeft().y();
3583 if (!subresDesc.sourceSize().isEmpty())
3584 size = subresDesc.sourceSize();
3585 size = clampedSubResourceUploadSize(size, dp, level, texD->m_pixelSize);
3586 if (size.width() == img.width()) {
3587 const int bpc = qMax(1, img.depth() / 8);
3588 Q_ASSERT(size.height() * img.bytesPerLine() <= fullImageSizeBytes);
3589 memcpy(reinterpret_cast<char *>(mp) + *curOfs,
3590 img.constBits() + sy * img.bytesPerLine() + sx * bpc,
3591 size.height() * img.bytesPerLine());
3592 } else {
3593 img = img.copy(sx, sy, size.width(), size.height());
3594 bpl = img.bytesPerLine();
3595 Q_ASSERT(img.sizeInBytes() <= fullImageSizeBytes);
3596 memcpy(reinterpret_cast<char *>(mp) + *curOfs, img.constBits(), size_t(img.sizeInBytes()));
3597 }
3598 } else {
3599 size = clampedSubResourceUploadSize(size, dp, level, texD->m_pixelSize);
3600 memcpy(reinterpret_cast<char *>(mp) + *curOfs, img.constBits(), size_t(fullImageSizeBytes));
3601 }
3602
3603 [blitEnc copyFromBuffer: texD->d->stagingBuf[currentFrameSlot]
3604 sourceOffset: NSUInteger(*curOfs)
3605 sourceBytesPerRow: NSUInteger(bpl)
3606 sourceBytesPerImage: 0
3607 sourceSize: MTLSizeMake(NSUInteger(size.width()), NSUInteger(size.height()), 1)
3608 toTexture: texD->d->tex
3609 destinationSlice: NSUInteger(is3D ? 0 : layer)
3610 destinationLevel: NSUInteger(level)
3611 destinationOrigin: MTLOriginMake(NSUInteger(dp.x()), NSUInteger(dp.y()), NSUInteger(is3D ? layer : 0))
3612 options: MTLBlitOptionNone];
3613
3614 *curOfs += aligned<qsizetype>(fullImageSizeBytes, QRhiMetalData::TEXBUF_ALIGN);
3615 } else if (!rawData.isEmpty() && isCompressedFormat(texD->m_format)) {
3616 const QSize subresSize = q->sizeForMipLevel(level, texD->m_pixelSize);
3617 const int subresw = subresSize.width();
3618 const int subresh = subresSize.height();
3619 int w, h;
3620 if (subresDesc.sourceSize().isEmpty()) {
3621 w = subresw;
3622 h = subresh;
3623 } else {
3624 w = subresDesc.sourceSize().width();
3625 h = subresDesc.sourceSize().height();
3626 }
3627
3628 quint32 bpl = 0;
3629 QSize blockDim;
3630 compressedFormatInfo(texD->m_format, QSize(w, h), &bpl, nullptr, &blockDim);
3631
3632 const int dx = aligned(dp.x(), blockDim.width());
3633 const int dy = aligned(dp.y(), blockDim.height());
3634 if (dx + w != subresw)
3635 w = aligned(w, blockDim.width());
3636 if (dy + h != subresh)
3637 h = aligned(h, blockDim.height());
3638
3639 memcpy(reinterpret_cast<char *>(mp) + *curOfs, rawData.constData(), size_t(rawData.size()));
3640
3641 [blitEnc copyFromBuffer: texD->d->stagingBuf[currentFrameSlot]
3642 sourceOffset: NSUInteger(*curOfs)
3643 sourceBytesPerRow: bpl
3644 sourceBytesPerImage: 0
3645 sourceSize: MTLSizeMake(NSUInteger(w), NSUInteger(h), 1)
3646 toTexture: texD->d->tex
3647 destinationSlice: NSUInteger(is3D ? 0 : layer)
3648 destinationLevel: NSUInteger(level)
3649 destinationOrigin: MTLOriginMake(NSUInteger(dx), NSUInteger(dy), NSUInteger(is3D ? layer : 0))
3650 options: MTLBlitOptionNone];
3651
3652 *curOfs += aligned<qsizetype>(rawData.size(), QRhiMetalData::TEXBUF_ALIGN);
3653 } else if (!rawData.isEmpty()) {
3654 const QSize subresSize = q->sizeForMipLevel(level, texD->m_pixelSize);
3655 const int subresw = subresSize.width();
3656 const int subresh = subresSize.height();
3657 int w, h;
3658 if (subresDesc.sourceSize().isEmpty()) {
3659 w = subresw;
3660 h = subresh;
3661 } else {
3662 w = subresDesc.sourceSize().width();
3663 h = subresDesc.sourceSize().height();
3664 }
3665
3666 QSize size = clampedSubResourceUploadSize(QSize(w, h), dp, level, texD->m_pixelSize);
3667 quint32 bytesPerPixel = 0;
3668 textureFormatInfo(texD->m_format, size, nullptr, nullptr, &bytesPerPixel);
3669 size = clampedSubResourceUploadSizeForSourceData(size, subresDesc.dataStride(),
3670 bytesPerPixel, rawData.size());
3671 w = size.width();
3672 h = size.height();
3673
3674 quint32 bpl = 0;
3675 if (subresDesc.dataStride())
3676 bpl = subresDesc.dataStride();
3677 else
3678 textureFormatInfo(texD->m_format, QSize(w, h), &bpl, nullptr, nullptr);
3679
3680 memcpy(reinterpret_cast<char *>(mp) + *curOfs, rawData.constData(), size_t(rawData.size()));
3681
3682 if (!size.isEmpty()) {
3683 [blitEnc copyFromBuffer: texD->d->stagingBuf[currentFrameSlot]
3684 sourceOffset: NSUInteger(*curOfs)
3685 sourceBytesPerRow: bpl
3686 sourceBytesPerImage: 0
3687 sourceSize: MTLSizeMake(NSUInteger(w), NSUInteger(h), 1)
3688 toTexture: texD->d->tex
3689 destinationSlice: NSUInteger(is3D ? 0 : layer)
3690 destinationLevel: NSUInteger(level)
3691 destinationOrigin: MTLOriginMake(NSUInteger(dp.x()), NSUInteger(dp.y()), NSUInteger(is3D ? layer : 0))
3692 options: MTLBlitOptionNone];
3693 }
3694
3695 *curOfs += aligned<qsizetype>(rawData.size(), QRhiMetalData::TEXBUF_ALIGN);
3696 } else {
3697 qWarning("Invalid texture upload for %p layer=%d mip=%d", texD, layer, level);
3698 }
3699}
3700
3701void QRhiMetal::enqueueResourceUpdates(QRhiCommandBuffer *cb, QRhiResourceUpdateBatch *resourceUpdates)
3702{
3703 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
3705
3706 id<MTLBlitCommandEncoder> blitEnc = nil;
3707 auto ensureBlit = [&blitEnc, cbD, this]() {
3708 if (!blitEnc) {
3709 blitEnc = [cbD->d->cb blitCommandEncoder];
3710 if (debugMarkers)
3711 [blitEnc pushDebugGroup: @"Resource updates"];
3712 }
3713 };
3714
3715 for (int opIdx = 0; opIdx < ud->activeBufferOpCount; ++opIdx) {
3716 const QRhiResourceUpdateBatchPrivate::BufferOp &u(ud->bufferOps[opIdx]);
3718 QMetalBuffer *bufD = QRHI_RES(QMetalBuffer, u.buf);
3719 Q_ASSERT(bufD->m_type == QRhiBuffer::Dynamic);
3720 for (int i = 0, ie = bufD->d->slotted ? QMTL_FRAMES_IN_FLIGHT : 1; i != ie; ++i) {
3721 if (u.offset == 0 && u.data.size() == bufD->m_size)
3722 bufD->d->pendingUpdates[i].clear();
3723 bufD->d->pendingUpdates[i].append({ u.offset, u.data });
3724 }
3726 QMetalBuffer *bufD = QRHI_RES(QMetalBuffer, u.buf);
3727 Q_ASSERT(bufD->m_type != QRhiBuffer::Dynamic);
3728 Q_ASSERT(u.offset + u.data.size() <= bufD->m_size);
3729 if (bufD->d->isDeviceLocal) {
3730 const quint32 size = quint32(u.data.size());
3731 if (size) { // a zero size blit is invalid
3732 quint32 stagingOffset = 0;
3733 id<MTLBuffer> stagingBuf = d->allocBufferStaging(size, currentFrameSlot, &stagingOffset);
3734 if (stagingBuf) {
3735 char *p = reinterpret_cast<char *>([stagingBuf contents]);
3736 memcpy(p + stagingOffset, u.data.constData(), size);
3737 ensureBlit();
3738 [blitEnc copyFromBuffer: stagingBuf
3739 sourceOffset: stagingOffset
3740 toBuffer: bufD->d->buf[0]
3741 destinationOffset: u.offset
3742 size: size];
3743 } else {
3744 qWarning("Failed to allocate staging buffer of size %u for buffer upload", size);
3745 }
3746 }
3747 bufD->lastActiveFrameSlot = currentFrameSlot;
3748 } else {
3749 for (int i = 0, ie = bufD->d->slotted ? QMTL_FRAMES_IN_FLIGHT : 1; i != ie; ++i)
3750 bufD->d->pendingUpdates[i].append({ u.offset, u.data });
3751 }
3753 QMetalBuffer *bufD = QRHI_RES(QMetalBuffer, u.buf);
3755 const int idx = bufD->d->slotted ? currentFrameSlot : 0;
3756 if (bufD->m_type == QRhiBuffer::Dynamic) {
3757 char *p = reinterpret_cast<char *>([bufD->d->buf[idx] contents]);
3758 if (p) {
3759 u.result->data.resize(u.readSize);
3760 memcpy(u.result->data.data(), p + u.offset, size_t(u.readSize));
3761 }
3762 if (u.result->completed)
3763 u.result->completed();
3764 } else {
3765 // Copy into a dedicated staging buffer, here and now, instead
3766 // of holding on to the buffer and reading it when the readback
3767 // completes. The contents have to be the ones at this point in
3768 // the command stream: a buffer that is not slotted - which is
3769 // every buffer with StorageBuffer usage - has one native buffer
3770 // shared by all frames, so by the time the readback completes a
3771 // later frame may well be writing it.
3772 QRhiMetalData::BufferReadback readback;
3773 readback.activeFrameSlot = currentFrameSlot;
3774 readback.readSize = u.readSize;
3775 readback.result = u.result;
3776 readback.buf = [d->dev newBufferWithLength: u.readSize
3777 options: MTLResourceStorageModeShared];
3778
3779 ensureBlit();
3780 [blitEnc copyFromBuffer: bufD->d->buf[idx]
3781 sourceOffset: u.offset
3782 toBuffer: readback.buf
3783 destinationOffset: 0
3784 size: u.readSize];
3785
3786 d->activeBufferReadbacks.append(readback);
3787 bufD->lastActiveFrameSlot = currentFrameSlot;
3788 }
3790 QMetalBuffer *dstD = QRHI_RES(QMetalBuffer, u.buf);
3791 QMetalBuffer *srcD = QRHI_RES(QMetalBuffer, u.src);
3792 Q_ASSERT(dstD->m_type != QRhiBuffer::Dynamic && srcD->m_type != QRhiBuffer::Dynamic);
3793
3795 const int srcIdx = srcD->d->slotted ? currentFrameSlot : 0;
3796 const int dstSlotCount = dstD->d->slotted ? QMTL_FRAMES_IN_FLIGHT : 1;
3797 // With device private buffers dstSlotCount is 1 and there are no
3798 // pending host writes at all, so this is a no-op. On the host
3799 // visible path every slot has to be flushed, not just the current
3800 // one: a partial copy cannot discard the pending writes outside the
3801 // copied range, and leaving them queued would let them land on top
3802 // of the copy in a later frame. Writing the slot that is not the
3803 // current one races a still in-flight frame reading it, which is
3804 // why mixing uploadStaticBuffer() and copyBuffer() on one buffer
3805 // within a frame stays unsupported there.
3806 for (int i = 0; i != dstSlotCount; ++i)
3808
3809 ensureBlit();
3810 for (int i = 0; i != dstSlotCount; ++i) {
3811 [blitEnc copyFromBuffer: srcD->d->buf[srcIdx]
3812 sourceOffset: u.srcOffset
3813 toBuffer: dstD->d->buf[i]
3814 destinationOffset: u.offset
3815 size: u.readSize];
3816 }
3817
3818 srcD->lastActiveFrameSlot = dstD->lastActiveFrameSlot = currentFrameSlot;
3820 QMetalBuffer *bufD = QRHI_RES(QMetalBuffer, u.buf);
3821 Q_ASSERT(bufD->m_type != QRhiBuffer::Dynamic && !bufD->d->slotted);
3823 ensureBlit();
3824 [blitEnc fillBuffer: bufD->d->buf[0]
3825 range: NSMakeRange(u.offset, u.readSize)
3826 value: u.fillValue];
3827 bufD->lastActiveFrameSlot = currentFrameSlot;
3828 }
3829 }
3830
3831 for (int opIdx = 0; opIdx < ud->activeTextureOpCount; ++opIdx) {
3832 const QRhiResourceUpdateBatchPrivate::TextureOp &u(ud->textureOps[opIdx]);
3834 QMetalTexture *utexD = QRHI_RES(QMetalTexture, u.dst);
3835 qsizetype stagingSize = 0;
3836 for (const auto &subres : u.subresDesc)
3837 stagingSize += subresUploadByteSize(subres.desc);
3838
3839 ensureBlit();
3840 Q_ASSERT(!utexD->d->stagingBuf[currentFrameSlot]);
3841 utexD->d->stagingBuf[currentFrameSlot] = [d->dev newBufferWithLength: NSUInteger(stagingSize)
3842 options: MTLResourceStorageModeShared];
3843
3844 void *mp = [utexD->d->stagingBuf[currentFrameSlot] contents];
3845 qsizetype curOfs = 0;
3846 for (const auto &subres : u.subresDesc)
3847 enqueueSubresUpload(utexD, mp, blitEnc, subres.layer, subres.level, subres.desc, &curOfs);
3848
3849 utexD->lastActiveFrameSlot = currentFrameSlot;
3850
3853 e.lastActiveFrameSlot = currentFrameSlot;
3854 e.stagingBuffer.buffer = utexD->d->stagingBuf[currentFrameSlot];
3855 utexD->d->stagingBuf[currentFrameSlot] = nil;
3856 d->releaseQueue.append(e);
3858 Q_ASSERT(u.src && u.dst);
3859 QMetalTexture *srcD = QRHI_RES(QMetalTexture, u.src);
3860 QMetalTexture *dstD = QRHI_RES(QMetalTexture, u.dst);
3861 const bool srcIs3D = srcD->m_flags.testFlag(QRhiTexture::ThreeDimensional);
3862 const bool dstIs3D = dstD->m_flags.testFlag(QRhiTexture::ThreeDimensional);
3863 const QPoint dp = u.desc.destinationTopLeft();
3864 const QSize mipSize = q->sizeForMipLevel(u.desc.sourceLevel(), srcD->m_pixelSize);
3865 const QSize copySize = u.desc.pixelSize().isEmpty() ? mipSize : u.desc.pixelSize();
3866 const QPoint sp = u.desc.sourceTopLeft();
3867
3868 ensureBlit();
3869 [blitEnc copyFromTexture: srcD->d->tex
3870 sourceSlice: NSUInteger(srcIs3D ? 0 : u.desc.sourceLayer())
3871 sourceLevel: NSUInteger(u.desc.sourceLevel())
3872 sourceOrigin: MTLOriginMake(NSUInteger(sp.x()), NSUInteger(sp.y()), NSUInteger(srcIs3D ? u.desc.sourceLayer() : 0))
3873 sourceSize: MTLSizeMake(NSUInteger(copySize.width()), NSUInteger(copySize.height()), 1)
3874 toTexture: dstD->d->tex
3875 destinationSlice: NSUInteger(dstIs3D ? 0 : u.desc.destinationLayer())
3876 destinationLevel: NSUInteger(u.desc.destinationLevel())
3877 destinationOrigin: MTLOriginMake(NSUInteger(dp.x()), NSUInteger(dp.y()), NSUInteger(dstIs3D ? u.desc.destinationLayer() : 0))];
3878
3879 srcD->lastActiveFrameSlot = dstD->lastActiveFrameSlot = currentFrameSlot;
3882 readback.activeFrameSlot = currentFrameSlot;
3883 readback.desc = u.rb;
3884 readback.result = u.result;
3885
3886 QMetalTexture *texD = QRHI_RES(QMetalTexture, u.rb.texture());
3887 QMetalSwapChain *swapChainD = nullptr;
3888 id<MTLTexture> src;
3889 QRect rect;
3890 bool is3D = false;
3891 if (texD) {
3892 if (texD->samples > 1) {
3893 qWarning("Multisample texture cannot be read back");
3894 continue;
3895 }
3896 is3D = texD->m_flags.testFlag(QRhiTexture::ThreeDimensional);
3897 if (u.rb.rect().isValid())
3898 rect = u.rb.rect();
3899 else
3900 rect = QRect({0, 0}, q->sizeForMipLevel(u.rb.level(), texD->m_pixelSize));
3901 readback.format = texD->m_format;
3902 src = texD->d->tex;
3903 texD->lastActiveFrameSlot = currentFrameSlot;
3904 } else {
3905 Q_ASSERT(currentSwapChain);
3907 if (u.rb.rect().isValid())
3908 rect = u.rb.rect();
3909 else
3910 rect = QRect({0, 0}, swapChainD->pixelSize);
3911 readback.format = swapChainD->d->rhiColorFormat;
3912 // Multisample swapchains need nothing special since resolving
3913 // happens when ending a renderpass.
3914 const QMetalRenderTargetData::ColorAtt &colorAtt(swapChainD->rtWrapper.d->fb.colorAtt[0]);
3915 src = colorAtt.resolveTex ? colorAtt.resolveTex : colorAtt.tex;
3916 }
3917 readback.pixelSize = rect.size();
3918
3919 quint32 bpl = 0;
3920 textureFormatInfo(readback.format, readback.pixelSize, &bpl, &readback.bufSize, nullptr);
3921 readback.buf = [d->dev newBufferWithLength: readback.bufSize options: MTLResourceStorageModeShared];
3922
3923 ensureBlit();
3924 [blitEnc copyFromTexture: src
3925 sourceSlice: NSUInteger(is3D ? 0 : u.rb.layer())
3926 sourceLevel: NSUInteger(u.rb.level())
3927 sourceOrigin: MTLOriginMake(NSUInteger(rect.x()), NSUInteger(rect.y()), NSUInteger(is3D ? u.rb.layer() : 0))
3928 sourceSize: MTLSizeMake(NSUInteger(rect.width()), NSUInteger(rect.height()), 1)
3929 toBuffer: readback.buf
3930 destinationOffset: 0
3931 destinationBytesPerRow: bpl
3932 destinationBytesPerImage: 0
3933 options: MTLBlitOptionNone];
3934
3935 d->activeTextureReadbacks.append(readback);
3937 QMetalTexture *utexD = QRHI_RES(QMetalTexture, u.dst);
3938 ensureBlit();
3939 [blitEnc generateMipmapsForTexture: utexD->d->tex];
3940 utexD->lastActiveFrameSlot = currentFrameSlot;
3941 }
3942 }
3943
3944 if (blitEnc) {
3945 if (debugMarkers)
3946 [blitEnc popDebugGroup];
3947 [blitEnc endEncoding];
3948 }
3949
3950 ud->free();
3951}
3952
3953// this handles all types of buffers, not just Dynamic
3955{
3956 if (bufD->d->pendingUpdates[slot].isEmpty())
3957 return;
3958
3959 Q_ASSERT(!bufD->d->isDeviceLocal);
3960
3961 void *p = [bufD->d->buf[slot] contents];
3962 quint32 changeBegin = UINT32_MAX;
3963 quint32 changeEnd = 0;
3964 for (const QMetalBufferData::BufferUpdate &u : std::as_const(bufD->d->pendingUpdates[slot])) {
3965 memcpy(static_cast<char *>(p) + u.offset, u.data.constData(), size_t(u.data.size()));
3966 if (u.offset < changeBegin)
3967 changeBegin = u.offset;
3968 if (u.offset + u.data.size() > changeEnd)
3969 changeEnd = u.offset + u.data.size();
3970 }
3971#ifdef Q_OS_MACOS
3972 if (changeBegin < UINT32_MAX && changeBegin < changeEnd && bufD->d->managed)
3973 [bufD->d->buf[slot] didModifyRange: NSMakeRange(NSUInteger(changeBegin), NSUInteger(changeEnd - changeBegin))];
3974#endif
3975
3976 bufD->d->pendingUpdates[slot].clear();
3977}
3978
3980{
3981 executeBufferHostWritesForSlot(bufD, bufD->d->slotted ? currentFrameSlot : 0);
3982}
3983
3984void QRhiMetal::resourceUpdate(QRhiCommandBuffer *cb, QRhiResourceUpdateBatch *resourceUpdates)
3985{
3986 Q_ASSERT(QRHI_RES(QMetalCommandBuffer, cb)->recordingPass == QMetalCommandBuffer::NoPass);
3987
3988 enqueueResourceUpdates(cb, resourceUpdates);
3989}
3990
3991void QRhiMetal::beginPass(QRhiCommandBuffer *cb,
3992 QRhiRenderTarget *rt,
3993 const QColor &colorClearValue,
3994 const QRhiDepthStencilClearValue &depthStencilClearValue,
3995 QRhiResourceUpdateBatch *resourceUpdates,
3997{
3998 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4000
4001 if (resourceUpdates)
4002 enqueueResourceUpdates(cb, resourceUpdates);
4003
4004 QMetalRenderTargetData *rtD = nullptr;
4005 switch (rt->resourceType()) {
4006 case QRhiResource::SwapChainRenderTarget:
4007 {
4009 rtD = rtSc->d;
4010 QRhiShadingRateMap *shadingRateMap = rtSc->swapChain()->shadingRateMap();
4011 cbD->d->currentPassRpDesc = d->createDefaultRenderPass(rtD->dsAttCount,
4012 colorClearValue,
4013 depthStencilClearValue,
4014 rtD->colorAttCount,
4015 shadingRateMap);
4016 if (rtD->colorAttCount) {
4017 QMetalRenderTargetData::ColorAtt &color0(rtD->fb.colorAtt[0]);
4019 Q_ASSERT(currentSwapChain);
4021 if (!swapChainD->d->curDrawable) {
4022 QMacAutoReleasePool pool;
4023 swapChainD->d->curDrawable = [[swapChainD->d->layer nextDrawable] retain];
4024 }
4025 if (!swapChainD->d->curDrawable) {
4026 qWarning("No drawable");
4027 return;
4028 }
4029 id<MTLTexture> scTex = swapChainD->d->curDrawable.texture;
4030 if (color0.needsDrawableForTex) {
4031 color0.tex = scTex;
4032 color0.needsDrawableForTex = false;
4033 } else {
4034 color0.resolveTex = scTex;
4035 color0.needsDrawableForResolveTex = false;
4036 }
4037 }
4038 }
4039 if (shadingRateMap)
4040 QRHI_RES(QMetalShadingRateMap, shadingRateMap)->lastActiveFrameSlot = currentFrameSlot;
4041 }
4042 break;
4043 case QRhiResource::TextureRenderTarget:
4044 {
4046 rtD = rtTex->d;
4047 if (!QRhiRenderTargetAttachmentTracker::isUpToDate<QMetalTexture, QMetalRenderBuffer>(rtTex->description(), rtD->currentResIdList))
4048 rtTex->create();
4049 cbD->d->currentPassRpDesc = d->createDefaultRenderPass(rtD->dsAttCount,
4050 colorClearValue,
4051 depthStencilClearValue,
4052 rtD->colorAttCount,
4053 rtTex->m_desc.shadingRateMap());
4054 if (rtD->fb.preserveColor) {
4055 for (uint i = 0; i < uint(rtD->colorAttCount); ++i)
4056 cbD->d->currentPassRpDesc.colorAttachments[i].loadAction = MTLLoadActionLoad;
4057 }
4058 if (rtD->dsAttCount && rtD->fb.preserveDs) {
4059 cbD->d->currentPassRpDesc.depthAttachment.loadAction = MTLLoadActionLoad;
4060 cbD->d->currentPassRpDesc.stencilAttachment.loadAction = MTLLoadActionLoad;
4061 }
4062 int colorAttCount = 0;
4063 for (auto it = rtTex->m_desc.cbeginColorAttachments(), itEnd = rtTex->m_desc.cendColorAttachments();
4064 it != itEnd; ++it)
4065 {
4066 colorAttCount += 1;
4067 if (it->texture()) {
4068 QRHI_RES(QMetalTexture, it->texture())->lastActiveFrameSlot = currentFrameSlot;
4069 if (it->multiViewCount() >= 2)
4070 cbD->d->currentPassRpDesc.renderTargetArrayLength = NSUInteger(it->multiViewCount());
4071 } else if (it->renderBuffer()) {
4072 QRHI_RES(QMetalRenderBuffer, it->renderBuffer())->lastActiveFrameSlot = currentFrameSlot;
4073 }
4074 if (it->resolveTexture())
4075 QRHI_RES(QMetalTexture, it->resolveTexture())->lastActiveFrameSlot = currentFrameSlot;
4076 }
4077 if (rtTex->m_desc.depthStencilBuffer())
4078 QRHI_RES(QMetalRenderBuffer, rtTex->m_desc.depthStencilBuffer())->lastActiveFrameSlot = currentFrameSlot;
4079 if (rtTex->m_desc.depthTexture()) {
4080 QMetalTexture *depthTexture = QRHI_RES(QMetalTexture, rtTex->m_desc.depthTexture());
4081 depthTexture->lastActiveFrameSlot = currentFrameSlot;
4082 if (depthTexture->arraySize() >= 2) {
4083 const int depthLayer = rtTex->m_desc.depthLayer();
4084 if (depthLayer >= 0) {
4085 cbD->d->currentPassRpDesc.depthAttachment.slice = NSUInteger(depthLayer);
4086 cbD->d->currentPassRpDesc.stencilAttachment.slice = NSUInteger(depthLayer);
4087 if (colorAttCount == 0)
4088 cbD->d->currentPassRpDesc.renderTargetArrayLength = 1;
4089 } else if (colorAttCount == 0) {
4090 cbD->d->currentPassRpDesc.renderTargetArrayLength = NSUInteger(depthTexture->arraySize());
4091 }
4092 }
4093 }
4094 if (rtTex->m_desc.depthResolveTexture())
4095 QRHI_RES(QMetalTexture, rtTex->m_desc.depthResolveTexture())->lastActiveFrameSlot = currentFrameSlot;
4096 if (rtTex->m_desc.shadingRateMap())
4097 QRHI_RES(QMetalShadingRateMap, rtTex->m_desc.shadingRateMap())->lastActiveFrameSlot = currentFrameSlot;
4098 }
4099 break;
4100 default:
4101 Q_UNREACHABLE();
4102 break;
4103 }
4104
4105 cbD->d->deferredColorStoreActions.clear();
4106 cbD->d->deferredDepthStoreAction = MTLStoreActionUnknown;
4107 cbD->d->deferredStencilStoreAction = MTLStoreActionUnknown;
4108 for (uint i = 0; i < uint(rtD->colorAttCount); ++i) {
4109 cbD->d->currentPassRpDesc.colorAttachments[i].texture = rtD->fb.colorAtt[i].tex;
4110 cbD->d->currentPassRpDesc.colorAttachments[i].slice = NSUInteger(rtD->fb.colorAtt[i].arrayLayer);
4111 cbD->d->currentPassRpDesc.colorAttachments[i].depthPlane = NSUInteger(rtD->fb.colorAtt[i].slice);
4112 cbD->d->currentPassRpDesc.colorAttachments[i].level = NSUInteger(rtD->fb.colorAtt[i].level);
4113 if (rtD->fb.colorAtt[i].resolveTex) {
4114 const MTLStoreAction storeAction = rtD->fb.preserveColor ? MTLStoreActionStoreAndMultisampleResolve
4115 : MTLStoreActionMultisampleResolve;
4116 // Defer, so that the multisample contents can be kept if the pass
4117 // ends up being interrupted. finalizeDeferredStoreActions() sets the
4118 // real action before the encoder ends.
4119 cbD->d->currentPassRpDesc.colorAttachments[i].storeAction = MTLStoreActionUnknown;
4120 cbD->d->deferredColorStoreActions.append({ i, storeAction });
4121 cbD->d->currentPassRpDesc.colorAttachments[i].resolveTexture = rtD->fb.colorAtt[i].resolveTex;
4122 cbD->d->currentPassRpDesc.colorAttachments[i].resolveSlice = NSUInteger(rtD->fb.colorAtt[i].resolveLayer);
4123 cbD->d->currentPassRpDesc.colorAttachments[i].resolveLevel = NSUInteger(rtD->fb.colorAtt[i].resolveLevel);
4124 }
4125 }
4126
4127 if (rtD->dsAttCount) {
4128 Q_ASSERT(rtD->fb.dsTex);
4129 cbD->d->currentPassRpDesc.depthAttachment.texture = rtD->fb.dsTex;
4130 cbD->d->currentPassRpDesc.stencilAttachment.texture = rtD->fb.hasStencil ? rtD->fb.dsTex : nil;
4131 if (rtD->fb.depthNeedsStore) { // Depth/Stencil is set to DontCare by default, override if needed
4132 cbD->d->currentPassRpDesc.depthAttachment.storeAction = MTLStoreActionStore;
4133 } else if (canStoreAttachment(rtD->fb.dsTex)) {
4134 // Would be discarded at the end of the pass, but an interruption in
4135 // the middle of it still has to be able to keep the contents. Defer,
4136 // so that nothing is stored unless that actually happens.
4137 cbD->d->currentPassRpDesc.depthAttachment.storeAction = MTLStoreActionUnknown;
4138 cbD->d->deferredDepthStoreAction = MTLStoreActionDontCare;
4139 if (rtD->fb.hasStencil) {
4140 cbD->d->currentPassRpDesc.stencilAttachment.storeAction = MTLStoreActionUnknown;
4141 cbD->d->deferredStencilStoreAction = MTLStoreActionDontCare;
4142 }
4143 }
4144 if (rtD->fb.dsResolveTex) {
4145 const MTLStoreAction dsStoreAction = rtD->fb.depthNeedsStore ? MTLStoreActionStoreAndMultisampleResolve
4146 : MTLStoreActionMultisampleResolve;
4147 // Deferred for the same reason as the color attachments above, but
4148 // only when there is something to defer to: a memoryless
4149 // depth-stencil buffer, which is what a QRhiRenderBuffer is on Apple
4150 // GPUs, cannot be stored at all, and rejects the store-and-resolve
4151 // action that an interruption would need.
4152 const bool deferrable = canStoreAttachment(rtD->fb.dsTex);
4153 cbD->d->currentPassRpDesc.depthAttachment.storeAction = deferrable ? MTLStoreActionUnknown
4154 : dsStoreAction;
4155 if (deferrable)
4156 cbD->d->deferredDepthStoreAction = dsStoreAction;
4157 cbD->d->currentPassRpDesc.depthAttachment.resolveTexture = rtD->fb.dsResolveTex;
4158 if (rtD->fb.hasStencil) {
4159 cbD->d->currentPassRpDesc.stencilAttachment.resolveTexture = rtD->fb.dsResolveTex;
4160 cbD->d->currentPassRpDesc.stencilAttachment.storeAction = deferrable ? MTLStoreActionUnknown
4161 : dsStoreAction;
4162 if (deferrable)
4163 cbD->d->deferredStencilStoreAction = dsStoreAction;
4164 }
4165 }
4166 }
4167
4168 cbD->d->currentRenderPassEncoder = [cbD->d->cb renderCommandEncoderWithDescriptor: cbD->d->currentPassRpDesc];
4169
4171
4173 cbD->currentTarget = rt;
4174}
4175
4176void QRhiMetal::endPass(QRhiCommandBuffer *cb, QRhiResourceUpdateBatch *resourceUpdates)
4177{
4178 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4180
4182 [cbD->d->currentRenderPassEncoder endEncoding];
4183
4185 cbD->currentTarget = nullptr;
4186
4187 if (resourceUpdates)
4188 enqueueResourceUpdates(cb, resourceUpdates);
4189}
4190
4191void QRhiMetal::beginComputePass(QRhiCommandBuffer *cb,
4192 QRhiResourceUpdateBatch *resourceUpdates,
4194{
4195 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4197
4198 if (resourceUpdates)
4199 enqueueResourceUpdates(cb, resourceUpdates);
4200
4201 cbD->d->currentComputePassEncoder = [cbD->d->cb computeCommandEncoder];
4204}
4205
4206void QRhiMetal::endComputePass(QRhiCommandBuffer *cb, QRhiResourceUpdateBatch *resourceUpdates)
4207{
4208 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4210
4211 [cbD->d->currentComputePassEncoder endEncoding];
4213
4214 if (resourceUpdates)
4215 enqueueResourceUpdates(cb, resourceUpdates);
4216}
4217
4218void QRhiMetal::setComputePipeline(QRhiCommandBuffer *cb, QRhiComputePipeline *ps)
4219{
4220 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4223
4225 cbD->currentGraphicsPipeline = nullptr;
4226 cbD->currentComputePipeline = psD;
4228
4229 [cbD->d->currentComputePassEncoder setComputePipelineState: psD->d->ps];
4230 }
4231
4232 psD->lastActiveFrameSlot = currentFrameSlot;
4233}
4234
4235void QRhiMetal::dispatch(QRhiCommandBuffer *cb, int x, int y, int z)
4236{
4237 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4240
4241 [cbD->d->currentComputePassEncoder dispatchThreadgroups: MTLSizeMake(NSUInteger(x), NSUInteger(y), NSUInteger(z))
4242 threadsPerThreadgroup: psD->d->localSize];
4243}
4244
4245void QRhiMetal::dispatchIndirect(QRhiCommandBuffer *cb, QRhiBuffer *indirectBuffer,
4246 quint32 indirectBufferOffset)
4247{
4248 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4251
4252 QMetalBuffer *indirectBufD = QRHI_RES(QMetalBuffer, indirectBuffer);
4254 indirectBufD->lastActiveFrameSlot = currentFrameSlot;
4255 id<MTLBuffer> indirectBufMtl = indirectBufD->d->buf[indirectBufD->d->slotted ? currentFrameSlot : 0];
4256
4257 // dispatchThreadgroups still wants threadsPerThreadgroup explicitly; the
4258 // indirect buffer only supplies the grid size in threadgroups. The
4259 // threads-per-threadgroup value comes from the bound compute pipeline
4260 // (set in QMetalComputePipeline::create() from the SPIR-V shader's
4261 // local_size_x/y/z layout qualifiers).
4262 [cbD->d->currentComputePassEncoder
4263 dispatchThreadgroupsWithIndirectBuffer: indirectBufMtl
4264 indirectBufferOffset: indirectBufferOffset
4265 threadsPerThreadgroup: psD->d->localSize];
4266}
4267
4268void QRhiMetal::drawIndirectCount(QRhiCommandBuffer *cb,
4269 QRhiBuffer *indirectBuffer, quint32 indirectBufferOffset,
4270 QRhiBuffer *countBuffer, quint32 countBufferOffset,
4271 quint32 maxDrawCount, quint32 stride)
4272{
4273 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4275
4276 // Implemented on top of indirect command buffers. There can be no CPU-side
4277 // fallback, the count is only known to the device.
4278 if (const char *reason = icbUnavailableReason(cbD)) {
4279 qWarning("drawIndirectCount is not available because %s; skipping", reason);
4280 return;
4281 }
4282
4283 icbDraw(cbD, false, QRHI_RES(QMetalBuffer, indirectBuffer), indirectBufferOffset,
4284 QRHI_RES(QMetalBuffer, countBuffer), countBufferOffset, maxDrawCount, stride);
4285}
4286
4287void QRhiMetal::drawIndexedIndirectCount(QRhiCommandBuffer *cb,
4288 QRhiBuffer *indirectBuffer, quint32 indirectBufferOffset,
4289 QRhiBuffer *countBuffer, quint32 countBufferOffset,
4290 quint32 maxDrawCount, quint32 stride)
4291{
4292 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4294
4295 // Implemented on top of indirect command buffers. There can be no CPU-side
4296 // fallback, the count is only known to the device.
4297 if (const char *reason = icbUnavailableReason(cbD)) {
4298 qWarning("drawIndexedIndirectCount is not available because %s; skipping", reason);
4299 return;
4300 }
4301
4302 if (!cbD->currentIndexBuffer) {
4303 qWarning("drawIndexedIndirectCount called without an index buffer bound; skipping");
4304 return;
4305 }
4306
4307 icbDraw(cbD, true, QRHI_RES(QMetalBuffer, indirectBuffer), indirectBufferOffset,
4308 QRHI_RES(QMetalBuffer, countBuffer), countBufferOffset, maxDrawCount, stride);
4309}
4310
4311static inline MTLPrimitiveType toMetalPrimitiveType(QRhiGraphicsPipeline::Topology t);
4312
4314{
4328
4329 enum class Fill { None, Cpu, Gpu };
4331 bool slotSetupFailed = false;
4332 bool created = false;
4333
4334 // Set by buildIndirect(), which makes the CPU-side recording irrelevant:
4335 // gpuFilled means the kernel encoded into frameSlots[builtSlot], fallbackBuild
4336 // that there was no ICB to encode into and executeIndirect() has to use the
4337 // plain indirect draw entry points.
4338 bool gpuFilled = false;
4339 bool fallbackBuild = false;
4340 QRhiIndirectCommandBufferBuildInfo buildInfo;
4341 int builtSlot = -1;
4342};
4343
4345{
4346 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
4347 QMetalIndirectCommandBufferData::Slot &slot(icbD->d->frameSlots[i]);
4348 if (slot.icb) {
4352 e.stagingIcbBuffer.icb = slot.icb;
4353 e.stagingIcbBuffer.argBuffer = slot.argBuffer;
4354 rhiD->d->releaseQueue.append(e);
4355 }
4356 if (slot.rangeBuffer) {
4360 e.stagingBuffer.buffer = slot.rangeBuffer;
4361 rhiD->d->releaseQueue.append(e);
4362 }
4363 slot = {};
4364 }
4366}
4367
4370{
4372
4373 if (icbD->d->slotSetupFailed)
4374 return false;
4375
4376 if (icbD->d->fill == fill)
4377 return icbD->d->frameSlots[0].icb != nil;
4378
4381 icbD->d->gpuFilled = false;
4382 icbD->d->builtSlot = -1;
4383 }
4384
4385 if (!rhiD->caps.indirectCommandBuffers)
4386 return false;
4387
4388 const bool gpu = fill == QMetalIndirectCommandBufferData::Fill::Gpu;
4389 if (gpu && !rhiD->prepareIcbKernels())
4390 return false;
4391
4392 MTLIndirectCommandBufferDescriptor *icbDesc = [MTLIndirectCommandBufferDescriptor new];
4393 icbDesc.commandTypes = icbD->type() == QRhiIndirectCommandBuffer::IndexedDraws
4394 ? MTLIndirectCommandTypeDrawIndexed : MTLIndirectCommandTypeDraw;
4395 // All state except the primitive type and the index buffer is inherited.
4396 icbDesc.inheritPipelineState = YES;
4397 icbDesc.inheritBuffers = YES;
4398 icbDesc.maxVertexBufferBindCount = 0;
4399 icbDesc.maxFragmentBufferBindCount = 0;
4400
4401 bool ok = true;
4402 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT && ok; ++i) {
4403 QMetalIndirectCommandBufferData::Slot &slot(icbD->d->frameSlots[i]);
4404 slot.icb = [rhiD->d->dev newIndirectCommandBufferWithDescriptor:icbDesc
4405 maxCommandCount:icbD->maxCommandCount()
4406 options:gpu ? MTLResourceStorageModePrivate
4407 : MTLResourceStorageModeShared];
4408 if (!slot.icb) {
4409 qWarning("Failed to create MTLIndirectCommandBuffer");
4410 ok = false;
4411 break;
4412 }
4413 if (gpu) {
4414 slot.rangeBuffer = [rhiD->d->dev newBufferWithLength:sizeof(MTLIndirectCommandBufferExecutionRange)
4415 options:MTLResourceStorageModePrivate];
4416 id<MTLArgumentEncoder> argEnc = [rhiD->d->icbEncodeFunction newArgumentEncoderWithBufferIndex:1];
4417 slot.argBuffer = [rhiD->d->dev newBufferWithLength:argEnc.encodedLength
4418 options:MTLResourceStorageModeShared];
4419 if (slot.rangeBuffer && slot.argBuffer) {
4420 [argEnc setArgumentBuffer:slot.argBuffer offset:0];
4421 [argEnc setIndirectCommandBuffer:slot.icb atIndex:0];
4422 } else {
4423 qWarning("Failed to create MTLIndirectCommandBuffer helper buffers");
4424 ok = false;
4425 }
4426 [argEnc release];
4427 if (!ok)
4428 break;
4429 }
4430 }
4431 [icbDesc release];
4432
4433 if (!ok) {
4434 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
4435 [icbD->d->frameSlots[i].icb release];
4436 [icbD->d->frameSlots[i].rangeBuffer release];
4437 [icbD->d->frameSlots[i].argBuffer release];
4438 icbD->d->frameSlots[i] = {};
4439 }
4440 icbD->d->slotSetupFailed = true;
4441 return false;
4442 }
4443
4444 icbD->d->fill = fill;
4445 return true;
4446}
4447
4449 quint32 maxCommandCount)
4452{
4453}
4454
4460
4462{
4463 clear();
4464
4465 if (!d->created)
4466 return;
4467
4468 d->slotSetupFailed = false;
4469 d->gpuFilled = false;
4470 d->fallbackBuild = false;
4471 d->buildInfo = {};
4472 d->builtSlot = -1;
4473 d->created = false;
4474 m_gpuBuilt = false;
4475 m_gpuBuiltCommandCount = 0;
4476
4477 QRHI_RES_RHI(QRhiMetal);
4478 if (rhiD) {
4480 rhiD->unregisterResource(this);
4481 }
4482}
4483
4485{
4486 if (d->created)
4487 destroy();
4488 else
4489 clear();
4490
4491 if (!m_maxCommandCount) {
4492 qWarning("QRhiIndirectCommandBuffer: maxCommandCount is 0");
4493 return false;
4494 }
4495
4496 QRHI_RES_RHI(QRhiMetal);
4498 d->created = true;
4499 rhiD->registerResource(this);
4500 return true;
4501}
4502
4503QRhiIndirectCommandBuffer *QRhiMetal::createIndirectCommandBuffer(QRhiIndirectCommandBuffer::Type type,
4504 quint32 maxCommandCount)
4505{
4506 return new QMetalIndirectCommandBuffer(this, type, maxCommandCount);
4507}
4508
4509void QRhiMetal::commitIndirectCommandBuffer(QRhiResourceUpdateBatch *u,
4510 QRhiIndirectCommandBuffer *icb)
4511{
4512 // Nothing to upload: executeIndirect() encodes straight into the ICB, and
4513 // only there are the primitive type and index buffer known.
4514 Q_UNUSED(u);
4515 Q_UNUSED(icb);
4516}
4517
4518// True when the slot holds an encoding that is already good for these
4519// contents and this state, and so does not need to be re-encoded.
4522 MTLPrimitiveType primitiveType,
4523 id<MTLBuffer> indexBufMtl, quint32 indexOffset,
4524 QRhiCommandBuffer::IndexFormat indexFormat)
4525{
4526 const bool indexed = icbD->type() == QRhiIndirectCommandBuffer::IndexedDraws;
4527
4528 return slot.encoded
4529 && slot.generation == icbD->contentsGeneration()
4530 && slot.primitiveType == primitiveType
4531 && (!indexed || (slot.indexBuf == indexBufMtl
4532 && slot.indexOffset == indexOffset
4533 && slot.indexFormat == indexFormat));
4534}
4535
4536// Re-encodes the CPU-recorded commands unless the slot already matches.
4539 MTLPrimitiveType primitiveType,
4540 id<MTLBuffer> indexBufMtl, quint32 indexOffset,
4541 QRhiCommandBuffer::IndexFormat indexFormat)
4542{
4543 const bool indexed = icbD->type() == QRhiIndirectCommandBuffer::IndexedDraws;
4544
4545 if (qrhimtl_icbSlotMatches(icbD, slot, primitiveType, indexBufMtl, indexOffset, indexFormat))
4546 return;
4547
4548 const quint32 count = icbD->recordedCommandCount();
4549 if (indexed) {
4550 const MTLIndexType indexType = indexFormat == QRhiCommandBuffer::IndexUInt16
4551 ? MTLIndexTypeUInt16 : MTLIndexTypeUInt32;
4552 const quint32 indexSize = indexFormat == QRhiCommandBuffer::IndexUInt16 ? 2 : 4;
4554 for (quint32 i = 0; i < count; ++i) {
4555 const QRhiIndexedIndirectDrawCommand &c(cmds[i]);
4556 id<MTLIndirectRenderCommand> rc = [slot.icb indirectRenderCommandAtIndex:i];
4557 [rc drawIndexedPrimitives:primitiveType
4558 indexCount:c.indexCount
4559 indexType:indexType
4560 indexBuffer:indexBufMtl
4561 indexBufferOffset:indexOffset + c.firstIndex * indexSize
4562 instanceCount:c.instanceCount
4563 baseVertex:c.vertexOffset
4564 baseInstance:c.firstInstance];
4565 }
4566 } else {
4567 const QRhiIndirectDrawCommand *cmds = icbD->drawCommands();
4568 for (quint32 i = 0; i < count; ++i) {
4569 const QRhiIndirectDrawCommand &c(cmds[i]);
4570 id<MTLIndirectRenderCommand> rc = [slot.icb indirectRenderCommandAtIndex:i];
4571 [rc drawPrimitives:primitiveType
4572 vertexStart:c.firstVertex
4573 vertexCount:c.vertexCount
4574 instanceCount:c.instanceCount
4575 baseInstance:c.firstInstance];
4576 }
4577 }
4578
4579 // Commands past the current count may be left over from a longer batch.
4580 if (count < icbD->maxCommandCount())
4581 [slot.icb resetWithRange:NSMakeRange(count, icbD->maxCommandCount() - count)];
4582
4583 slot.generation = icbD->contentsGeneration();
4584 slot.primitiveType = primitiveType;
4585 slot.indexBuf = indexBufMtl;
4586 slot.indexOffset = indexOffset;
4587 slot.indexFormat = indexFormat;
4588 slot.encoded = true;
4589}
4590
4591// Fallback for when there is no usable ICB.
4593 quint32 firstCommand, quint32 count,
4594 int currentFrameSlot)
4595{
4596 const MTLPrimitiveType primitiveType = cbD->currentGraphicsPipeline->d->primitiveType;
4597
4598 if (icbD->type() == QRhiIndirectCommandBuffer::IndexedDraws) {
4599 QMetalBuffer *indexBufD = cbD->currentIndexBuffer;
4600 if (!indexBufD)
4601 return;
4602 id<MTLBuffer> indexBufMtl = indexBufD->d->buf[indexBufD->d->slotted ? currentFrameSlot : 0];
4603 const MTLIndexType indexType = cbD->currentIndexFormat == QRhiCommandBuffer::IndexUInt16
4604 ? MTLIndexTypeUInt16 : MTLIndexTypeUInt32;
4605 const quint32 indexSize = cbD->currentIndexFormat == QRhiCommandBuffer::IndexUInt16 ? 2 : 4;
4607 for (quint32 i = 0; i < count; ++i) {
4608 const QRhiIndexedIndirectDrawCommand &c(cmds[firstCommand + i]);
4609 [cbD->d->currentRenderPassEncoder drawIndexedPrimitives: primitiveType
4610 indexCount: c.indexCount
4611 indexType: indexType
4612 indexBuffer: indexBufMtl
4613 indexBufferOffset: cbD->currentIndexOffset + c.firstIndex * indexSize
4614 instanceCount: c.instanceCount
4615 baseVertex: c.vertexOffset
4616 baseInstance: c.firstInstance];
4617 }
4618 } else {
4619 const QRhiIndirectDrawCommand *cmds = icbD->drawCommands();
4620 for (quint32 i = 0; i < count; ++i) {
4621 const QRhiIndirectDrawCommand &c(cmds[firstCommand + i]);
4622 [cbD->d->currentRenderPassEncoder drawPrimitives: primitiveType
4623 vertexStart: c.firstVertex
4624 vertexCount: c.vertexCount
4625 instanceCount: c.instanceCount
4626 baseInstance: c.firstInstance];
4627 }
4628 }
4629}
4630
4631void QRhiMetal::executeIndirect(QRhiCommandBuffer *cb, QRhiIndirectCommandBuffer *icb,
4632 quint32 firstCommand, quint32 commandCount)
4633{
4634 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4636
4637 QMetalIndirectCommandBuffer *icbD = QRHI_RES(QMetalIndirectCommandBuffer, icb);
4638 icbD->lastActiveFrameSlot = currentFrameSlot;
4639
4640 const bool indexed = icb->type() == QRhiIndirectCommandBuffer::IndexedDraws;
4641
4642 if (icbD->d->gpuFilled) {
4643 if (icbD->d->buildInfo.topology != cbD->currentGraphicsPipeline->topology()) {
4644 qWarning("executeIndirect: the indirect command buffer was built for a different "
4645 "topology than the current graphics pipeline uses; skipping");
4646 return;
4647 }
4648 QMetalIndirectCommandBufferData::Slot &slot(icbD->d->frameSlots[icbD->d->builtSlot]);
4649
4650 const quint32 total = icbD->commandCount();
4651 const bool deviceCount = icbD->d->buildInfo.countBuffer != nullptr;
4652 NSRange range = NSMakeRange(0, total);
4653 if (deviceCount) {
4654 // The count is only known to the device, and applying it is what
4655 // the compute kernel-written range is for. That form of
4656 // executeCommandsInBuffer takes no CPU-side range, so a subrange
4657 // and a device-side count cannot be combined.
4658 if (firstCommand != 0 || commandCount < total) {
4659 qWarning("executeIndirect: firstCommand and commandCount cannot be honoured "
4660 "together with a device-side count; executing all %u command(s)", total);
4661 }
4662 } else {
4663 if (firstCommand >= total)
4664 return;
4665 const quint32 count = qMin(commandCount, total - firstCommand);
4666 if (!count)
4667 return;
4668 range = NSMakeRange(firstCommand, count);
4669 }
4670
4671 if (indexed && icbD->d->buildInfo.indexBuffer) {
4672 QMetalBuffer *indexBufD = QRHI_RES(QMetalBuffer, icbD->d->buildInfo.indexBuffer);
4673 id<MTLBuffer> indexBufMtl = indexBufD->d->buf[indexBufD->d->slotted ? currentFrameSlot : 0];
4674 [cbD->d->currentRenderPassEncoder useResource:indexBufMtl
4675 usage:MTLResourceUsageRead
4676 stages:MTLRenderStageVertex | MTLRenderStageFragment];
4677 }
4678 if (deviceCount) {
4679 [cbD->d->currentRenderPassEncoder executeCommandsInBuffer:slot.icb
4680 indirectBuffer:slot.rangeBuffer
4681 indirectBufferOffset:0];
4682 } else {
4683 [cbD->d->currentRenderPassEncoder executeCommandsInBuffer:slot.icb withRange:range];
4684 }
4685 return;
4686 }
4687
4688 if (icbD->d->fallbackBuild) {
4689 const QRhiIndirectCommandBufferBuildInfo &info(icbD->d->buildInfo);
4690 if (info.countBuffer) {
4691 qWarning("executeIndirect: a device-side count needs an indirect command buffer, "
4692 "which is not available here; skipping");
4693 return;
4694 }
4695 const quint32 canonicalStride = indexed ? sizeof(QRhiIndexedIndirectDrawCommand)
4696 : sizeof(QRhiIndirectDrawCommand);
4697 const quint32 stride = info.stride ? info.stride : canonicalStride;
4698 const quint32 total = icbD->commandCount();
4699 if (firstCommand >= total)
4700 return;
4701 const quint32 count = qMin(commandCount, total - firstCommand);
4702 if (!count)
4703 return;
4704 const quint32 offset = info.sourceBufferOffset + firstCommand * stride;
4705 if (indexed)
4706 drawIndexedIndirect(cb, info.sourceBuffer, offset, count, stride);
4707 else
4708 drawIndirect(cb, info.sourceBuffer, offset, count, stride);
4709 return;
4710 }
4711
4712 const quint32 total = icbD->recordedCommandCount();
4713 if (firstCommand >= total)
4714 return;
4715 const quint32 count = qMin(commandCount, total - firstCommand);
4716 if (!count)
4717 return;
4718
4721 {
4722 qrhimtl_replayIcbOnCpu(cbD, icbD, firstCommand, count, currentFrameSlot);
4723 return;
4724 }
4725 QMetalIndirectCommandBufferData::Slot &slot(icbD->d->frameSlots[currentFrameSlot]);
4726
4727 id<MTLBuffer> indexBufMtl = nil;
4728 quint32 indexOffset = 0;
4729 if (indexed) {
4730 QMetalBuffer *indexBufD = cbD->currentIndexBuffer;
4731 if (!indexBufD)
4732 return;
4733 indexBufD->lastActiveFrameSlot = currentFrameSlot;
4734 indexBufMtl = indexBufD->d->buf[indexBufD->d->slotted ? currentFrameSlot : 0];
4735 indexOffset = cbD->currentIndexOffset;
4736 }
4737
4738 const MTLPrimitiveType primitiveType = cbD->currentGraphicsPipeline->d->primitiveType;
4739 if (slot.usedInFrameId == d->globalFrameId
4740 && !qrhimtl_icbSlotMatches(icbD, slot, primitiveType, indexBufMtl, indexOffset,
4741 cbD->currentIndexFormat))
4742 {
4743 // An executeCommandsInBuffer recorded earlier in this frame references
4744 // this slot, and re-encoding it now would change what that one executes
4745 // once submitted. Replaying as ordinary draws keeps both correct.
4746 qWarning("executeIndirect: the same indirect command buffer is executed more than once "
4747 "in a frame, with different contents, topology or index buffer state; "
4748 "falling back to individual draw calls");
4749 qrhimtl_replayIcbOnCpu(cbD, icbD, firstCommand, count, currentFrameSlot);
4750 return;
4751 }
4752
4753 qrhimtl_encodeIcbFromCpu(icbD, slot, primitiveType,
4754 indexBufMtl, indexOffset, cbD->currentIndexFormat);
4755
4756 if (indexed) {
4757 // Index buffer is not inherited from the encoder, and the ICB commands reference it directly,
4758 // so it needs the useResource.
4759 [cbD->d->currentRenderPassEncoder useResource:indexBufMtl
4760 usage:MTLResourceUsageRead
4761 stages:MTLRenderStageVertex | MTLRenderStageFragment];
4762 }
4763 [cbD->d->currentRenderPassEncoder executeCommandsInBuffer:slot.icb
4764 withRange:NSMakeRange(firstCommand, count)];
4765 slot.usedInFrameId = d->globalFrameId;
4766}
4767
4768void QRhiMetal::buildIndirect(QRhiCommandBuffer *cb, QRhiIndirectCommandBuffer *icb,
4769 const QRhiIndirectCommandBufferBuildInfo &info)
4770{
4771 QMetalCommandBuffer *cbD = QRHI_RES(QMetalCommandBuffer, cb);
4773
4774 QMetalIndirectCommandBuffer *icbD = QRHI_RES(QMetalIndirectCommandBuffer, icb);
4775 icbD->lastActiveFrameSlot = currentFrameSlot;
4776
4777 const bool indexed = icb->type() == QRhiIndirectCommandBuffer::IndexedDraws;
4778
4779 if (indexed && !info.indexBuffer) {
4780 qWarning("buildIndirect: an IndexedDraws indirect command buffer needs an "
4781 "index buffer in QRhiIndirectCommandBufferBuildInfo; skipping");
4782 return;
4783 }
4784
4785 quint32 count = info.commandCount ? info.commandCount : icbD->m_maxCommandCount;
4786 if (count > icbD->m_maxCommandCount) {
4787 qWarning("QRhiIndirectCommandBuffer: buildIndirect() with commandCount %u exceeds "
4788 "maxCommandCount %u; clamping", count, icbD->m_maxCommandCount);
4789 count = icbD->m_maxCommandCount;
4790 }
4791 icbD->m_gpuBuilt = true;
4792 icbD->m_gpuBuiltCommandCount = count;
4793
4795 icbD->d->gpuFilled = false;
4796 icbD->d->fallbackBuild = true;
4797 icbD->d->buildInfo = info;
4798 icbD->d->builtSlot = -1;
4799 return;
4800 }
4801
4802 QMetalIndirectCommandBufferData::Slot &slot(icbD->d->frameSlots[currentFrameSlot]);
4803
4804 QMetalBuffer *srcBufD = QRHI_RES(QMetalBuffer, info.sourceBuffer);
4806 srcBufD->lastActiveFrameSlot = currentFrameSlot;
4807 id<MTLBuffer> srcBufMtl = srcBufD->d->buf[srcBufD->d->slotted ? currentFrameSlot : 0];
4808
4809 id<MTLBuffer> indexBufMtl = nil;
4810 if (indexed) {
4811 QMetalBuffer *indexBufD = QRHI_RES(QMetalBuffer, info.indexBuffer);
4812 indexBufD->lastActiveFrameSlot = currentFrameSlot;
4813 indexBufMtl = indexBufD->d->buf[indexBufD->d->slotted ? currentFrameSlot : 0];
4814 }
4815
4816 id<MTLBuffer> countBufMtl = nil;
4817 if (info.countBuffer) {
4818 QMetalBuffer *countBufD = QRHI_RES(QMetalBuffer, info.countBuffer);
4820 countBufD->lastActiveFrameSlot = currentFrameSlot;
4821 countBufMtl = countBufD->d->buf[countBufD->d->slotted ? currentFrameSlot : 0];
4822 }
4823
4824 const quint32 stride = info.stride ? info.stride
4825 : (indexed ? sizeof(QRhiIndexedIndirectDrawCommand)
4826 : sizeof(QRhiIndirectDrawCommand));
4827
4828 // Outside a render pass, so the compute encoder costs no interruption, no
4829 // store action juggling and no per-pass state to restore.
4830 id<MTLComputeCommandEncoder> computeEncoder = [cbD->d->cb computeCommandEncoder];
4831 encodeIcbWithCompute(d, computeEncoder, slot.icb, slot.argBuffer, slot.rangeBuffer,
4832 indexed, info.indexFormat,
4833 toMetalPrimitiveType(info.topology),
4834 srcBufMtl, info.sourceBufferOffset,
4835 indexBufMtl, info.indexBufferOffset,
4836 countBufMtl, info.countBufferOffset,
4837 icbD->commandCount(), stride);
4838 [computeEncoder endEncoding];
4839
4840 icbD->d->gpuFilled = true;
4841 icbD->d->fallbackBuild = false;
4842 icbD->d->buildInfo = info;
4843 icbD->d->builtSlot = currentFrameSlot;
4844 // The CPU-encoded contents of this slot, if any, are gone now.
4845 slot.encoded = false;
4846}
4847
4849{
4850 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i)
4851 [e.buffer.buffers[i] release];
4852}
4853
4855{
4856 [e.renderbuffer.texture release];
4857}
4858
4860{
4861 [e.texture.texture release];
4862 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i)
4863 [e.texture.stagingBuffers[i] release];
4864 for (int i = 0; i < QRhi::MAX_MIP_LEVELS; ++i)
4865 [e.texture.views[i] release];
4866 [e.texture.samplingView release];
4867 [e.texture.writeView release];
4868}
4869
4871{
4872 [e.sampler.samplerState release];
4873}
4874
4876{
4877 qsizetype keepBegin = d->releaseQueue.size();
4878 for (qsizetype i = keepBegin - 1; i >= 0; --i) {
4879 const QRhiMetalData::DeferredReleaseEntry &e(d->releaseQueue[i]);
4880 if (forced || currentFrameSlot == e.lastActiveFrameSlot || e.lastActiveFrameSlot < 0) {
4881 switch (e.type) {
4884 break;
4887 break;
4890 break;
4893 break;
4894 case QRhiMetalData::DeferredReleaseEntry::StagingBuffer:
4895 [e.stagingBuffer.buffer release];
4896 break;
4897 case QRhiMetalData::DeferredReleaseEntry::GraphicsPipeline:
4898 [e.graphicsPipeline.pipelineState release];
4899 [e.graphicsPipeline.depthStencilState release];
4900 [e.graphicsPipeline.tessVertexComputeState[0] release];
4901 [e.graphicsPipeline.tessVertexComputeState[1] release];
4902 [e.graphicsPipeline.tessVertexComputeState[2] release];
4903 [e.graphicsPipeline.tessTessControlComputeState release];
4904 break;
4905 case QRhiMetalData::DeferredReleaseEntry::ComputePipeline:
4906 [e.computePipeline.pipelineState release];
4907 break;
4908 case QRhiMetalData::DeferredReleaseEntry::ShadingRateMap:
4909 [e.shadingRateMap.rateMap release];
4910 break;
4911 case QRhiMetalData::DeferredReleaseEntry::StagingIcbBuffer:
4912 [e.stagingIcbBuffer.icb release];
4913 [e.stagingIcbBuffer.argBuffer release];
4914 break;
4915 default:
4916 break;
4917 }
4918 } else if (--keepBegin != i) {
4919 d->releaseQueue[keepBegin] = std::move(d->releaseQueue[i]);
4920 }
4921 }
4922 if (keepBegin)
4923 d->releaseQueue.remove(0, keepBegin);
4924}
4925
4927{
4928 QVarLengthArray<std::function<void()>, 4> completedCallbacks;
4929
4930 for (int i = d->activeTextureReadbacks.count() - 1; i >= 0; --i) {
4931 const QRhiMetalData::TextureReadback &readback(d->activeTextureReadbacks[i]);
4932 if (forced || currentFrameSlot == readback.activeFrameSlot || readback.activeFrameSlot < 0) {
4933 readback.result->format = readback.format;
4934 readback.result->pixelSize = readback.pixelSize;
4935 readback.result->data.resize(int(readback.bufSize));
4936 void *p = [readback.buf contents];
4937 memcpy(readback.result->data.data(), p, readback.bufSize);
4938 [readback.buf release];
4939
4940 if (readback.result->completed)
4941 completedCallbacks.append(readback.result->completed);
4942
4943 d->activeTextureReadbacks.remove(i);
4944 }
4945 }
4946
4947 for (int i = d->activeBufferReadbacks.count() - 1; i >= 0; --i) {
4948 const QRhiMetalData::BufferReadback &readback(d->activeBufferReadbacks[i]);
4949 if (forced || currentFrameSlot == readback.activeFrameSlot
4950 || readback.activeFrameSlot < 0) {
4951 readback.result->data.resize(readback.readSize);
4952 char *p = reinterpret_cast<char *>([readback.buf contents]);
4953 Q_ASSERT(p);
4954 memcpy(readback.result->data.data(), p, size_t(readback.readSize));
4955 [readback.buf release];
4956
4957 if (readback.result->completed)
4958 completedCallbacks.append(readback.result->completed);
4959
4960 d->activeBufferReadbacks.remove(i);
4961 }
4962 }
4963
4964 for (auto f : completedCallbacks)
4965 f();
4966}
4967
4968QMetalBuffer::QMetalBuffer(QRhiImplementation *rhi, Type type, UsageFlags usage, quint32 size)
4970 d(new QMetalBufferData)
4971{
4972 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i)
4973 d->buf[i] = nil;
4974}
4975
4977{
4978 destroy();
4979 delete d;
4980}
4981
4983{
4984 if (!d->buf[0])
4985 return;
4986
4990
4991 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
4992 e.buffer.buffers[i] = d->buf[i];
4993 d->buf[i] = nil;
4994 d->pendingUpdates[i].clear();
4995 }
4996
4997 QRHI_RES_RHI(QRhiMetal);
4998 if (rhiD) {
4999 rhiD->d->releaseQueue.append(e);
5000 rhiD->unregisterResource(this);
5001 }
5002}
5003
5005{
5006 if (d->buf[0])
5007 destroy();
5008
5009 if (m_usage.testFlag(QRhiBuffer::StorageBuffer) && m_type == Dynamic) {
5010 qWarning("StorageBuffer cannot be combined with Dynamic");
5011 return false;
5012 }
5013
5014 const quint32 nonZeroSize = m_size <= 0 ? 256 : m_size;
5015 const quint32 roundedSize = m_usage.testFlag(QRhiBuffer::UniformBuffer) ? aligned(nonZeroSize, 256u) : nonZeroSize;
5016
5017 d->managed = false;
5018 d->isDeviceLocal = false;
5019 MTLResourceOptions opts = MTLResourceStorageModeShared;
5020
5021 QRHI_RES_RHI(QRhiMetal);
5022
5023 const bool internalHostWritable = (int(m_usage) & (WorkBufPoolUsage | InternalHostWritable)) != 0;
5024
5025 if (m_type != Dynamic && !internalHostWritable && rhiD->caps.usePrivateStaticBuffers) {
5026 opts = MTLResourceStorageModePrivate;
5027 d->isDeviceLocal = true;
5028 d->slotted = false;
5029 } else {
5030#ifdef Q_OS_MACOS
5031 if (!rhiD->caps.isAppleGPU && m_type != Dynamic) {
5032 opts = MTLResourceStorageModeManaged;
5033 d->managed = true;
5034 }
5035#endif
5036 // Have QMTL_FRAMES_IN_FLIGHT versions regardless of the type. This is
5037 // because writing to a host visible buffer is not safe when another
5038 // frame reading from the same buffer is still in flight.
5039 d->slotted = !m_usage.testFlag(QRhiBuffer::StorageBuffer) // except for SSBOs written in the shader
5040 && !internalHostWritable; // and the internal buffers
5041 }
5042
5043 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
5044 if (i == 0 || d->slotted) {
5045 d->buf[i] = [rhiD->d->dev newBufferWithLength: roundedSize options: opts];
5046 if (!m_objectName.isEmpty()) {
5047 if (!d->slotted) {
5048 d->buf[i].label = [NSString stringWithUTF8String: m_objectName.constData()];
5049 } else {
5050 const QByteArray name = m_objectName + '/' + QByteArray::number(i);
5051 d->buf[i].label = [NSString stringWithUTF8String: name.constData()];
5052 }
5053 }
5054 }
5055 }
5056
5058 generation += 1;
5059 rhiD->registerResource(this);
5060 return true;
5061}
5062
5064{
5065 if (d->slotted) {
5066 NativeBuffer b;
5067 Q_ASSERT(sizeof(b.objects) / sizeof(b.objects[0]) >= size_t(QMTL_FRAMES_IN_FLIGHT));
5068 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
5069 QRHI_RES_RHI(QRhiMetal);
5071 b.objects[i] = &d->buf[i];
5072 }
5073 b.slotCount = QMTL_FRAMES_IN_FLIGHT;
5074 return b;
5075 }
5076 if (!d->isDeviceLocal) {
5077 QRHI_RES_RHI(QRhiMetal);
5079 }
5080 return { { &d->buf[0] }, 1 };
5081}
5082
5084{
5085 // Shortcut the entire buffer update mechanism and allow the client to do
5086 // the host writes directly to the buffer. This will lead to unexpected
5087 // results when combined with QRhiResourceUpdateBatch-based updates for the
5088 // buffer, but provides a fast path for dynamic buffers that have all their
5089 // content changed in every frame.
5090 Q_ASSERT(m_type == Dynamic);
5091 Q_ASSERT(!d->isDeviceLocal);
5092 QRHI_RES_RHI(QRhiMetal);
5093 Q_ASSERT(rhiD->inFrame);
5094 const int slot = d->slotted ? rhiD->currentFrameSlot : 0;
5095 void *p = [d->buf[slot] contents];
5096 return static_cast<char *>(p);
5097}
5098
5100{
5101 // Nothing to do: managed is never set for Dynamic buffers, and so there is
5102 // no didModifyRange: to issue.
5103}
5104
5105static inline MTLPixelFormat toMetalTextureFormat(QRhiTexture::Format format, QRhiTexture::Flags flags, const QRhiMetal *d)
5106{
5107#ifndef Q_OS_MACOS
5108 Q_UNUSED(d);
5109#endif
5110
5111 const bool srgb = flags.testFlag(QRhiTexture::sRGB);
5112 switch (format) {
5113 case QRhiTexture::RGBA8:
5114 return srgb ? MTLPixelFormatRGBA8Unorm_sRGB : MTLPixelFormatRGBA8Unorm;
5115 case QRhiTexture::BGRA8:
5116 return srgb ? MTLPixelFormatBGRA8Unorm_sRGB : MTLPixelFormatBGRA8Unorm;
5117 case QRhiTexture::R8:
5118#ifdef Q_OS_MACOS
5119 return MTLPixelFormatR8Unorm;
5120#else
5121 return srgb ? MTLPixelFormatR8Unorm_sRGB : MTLPixelFormatR8Unorm;
5122#endif
5123 case QRhiTexture::R8SI:
5124 return MTLPixelFormatR8Sint;
5125 case QRhiTexture::R8UI:
5126 return MTLPixelFormatR8Uint;
5127 case QRhiTexture::RG8:
5128#ifdef Q_OS_MACOS
5129 return MTLPixelFormatRG8Unorm;
5130#else
5131 return srgb ? MTLPixelFormatRG8Unorm_sRGB : MTLPixelFormatRG8Unorm;
5132#endif
5133 case QRhiTexture::R16:
5134 return MTLPixelFormatR16Unorm;
5135 case QRhiTexture::RG16:
5136 return MTLPixelFormatRG16Unorm;
5137 case QRhiTexture::RED_OR_ALPHA8:
5138 return MTLPixelFormatR8Unorm;
5139
5140 case QRhiTexture::RGBA16F:
5141 return MTLPixelFormatRGBA16Float;
5142 case QRhiTexture::RGBA32F:
5143 return MTLPixelFormatRGBA32Float;
5144 case QRhiTexture::R16F:
5145 return MTLPixelFormatR16Float;
5146 case QRhiTexture::R32F:
5147 return MTLPixelFormatR32Float;
5148
5149 case QRhiTexture::RGB10A2:
5150 return MTLPixelFormatRGB10A2Unorm;
5151
5152 case QRhiTexture::R32SI:
5153 return MTLPixelFormatR32Sint;
5154 case QRhiTexture::R32UI:
5155 return MTLPixelFormatR32Uint;
5156 case QRhiTexture::RG32SI:
5157 return MTLPixelFormatRG32Sint;
5158 case QRhiTexture::RG32UI:
5159 return MTLPixelFormatRG32Uint;
5160 case QRhiTexture::RGBA32SI:
5161 return MTLPixelFormatRGBA32Sint;
5162 case QRhiTexture::RGBA32UI:
5163 return MTLPixelFormatRGBA32Uint;
5164
5165#ifdef Q_OS_MACOS
5166 case QRhiTexture::D16:
5167 return MTLPixelFormatDepth16Unorm;
5168 case QRhiTexture::D24:
5169 return [d->d->dev isDepth24Stencil8PixelFormatSupported] ? MTLPixelFormatDepth24Unorm_Stencil8 : MTLPixelFormatDepth32Float;
5170 case QRhiTexture::D24S8:
5171 return [d->d->dev isDepth24Stencil8PixelFormatSupported] ? MTLPixelFormatDepth24Unorm_Stencil8 : MTLPixelFormatDepth32Float_Stencil8;
5172#else
5173 case QRhiTexture::D16:
5174 return MTLPixelFormatDepth32Float;
5175 case QRhiTexture::D24:
5176 return MTLPixelFormatDepth32Float;
5177 case QRhiTexture::D24S8:
5178 return MTLPixelFormatDepth32Float_Stencil8;
5179#endif
5180 case QRhiTexture::D32F:
5181 return MTLPixelFormatDepth32Float;
5182 case QRhiTexture::D32FS8:
5183 return MTLPixelFormatDepth32Float_Stencil8;
5184
5185#ifdef Q_OS_MACOS
5186 case QRhiTexture::BC1:
5187 return srgb ? MTLPixelFormatBC1_RGBA_sRGB : MTLPixelFormatBC1_RGBA;
5188 case QRhiTexture::BC2:
5189 return srgb ? MTLPixelFormatBC2_RGBA_sRGB : MTLPixelFormatBC2_RGBA;
5190 case QRhiTexture::BC3:
5191 return srgb ? MTLPixelFormatBC3_RGBA_sRGB : MTLPixelFormatBC3_RGBA;
5192 case QRhiTexture::BC4:
5193 return MTLPixelFormatBC4_RUnorm;
5194 case QRhiTexture::BC5:
5195 qWarning("QRhiMetal does not support BC5");
5196 return MTLPixelFormatInvalid;
5197 case QRhiTexture::BC6H:
5198 return MTLPixelFormatBC6H_RGBUfloat;
5199 case QRhiTexture::BC7:
5200 return srgb ? MTLPixelFormatBC7_RGBAUnorm_sRGB : MTLPixelFormatBC7_RGBAUnorm;
5201#else
5202 case QRhiTexture::BC1:
5203 case QRhiTexture::BC2:
5204 case QRhiTexture::BC3:
5205 case QRhiTexture::BC4:
5206 case QRhiTexture::BC5:
5207 case QRhiTexture::BC6H:
5208 case QRhiTexture::BC7:
5209 qWarning("QRhiMetal: BCx compression not supported on this platform");
5210 return MTLPixelFormatInvalid;
5211#endif
5212
5213#ifndef Q_OS_MACOS
5214 case QRhiTexture::ETC2_RGB8:
5215 return srgb ? MTLPixelFormatETC2_RGB8_sRGB : MTLPixelFormatETC2_RGB8;
5216 case QRhiTexture::ETC2_RGB8A1:
5217 return srgb ? MTLPixelFormatETC2_RGB8A1_sRGB : MTLPixelFormatETC2_RGB8A1;
5218 case QRhiTexture::ETC2_RGBA8:
5219 return srgb ? MTLPixelFormatEAC_RGBA8_sRGB : MTLPixelFormatEAC_RGBA8;
5220
5221 case QRhiTexture::ASTC_4x4:
5222 return srgb ? MTLPixelFormatASTC_4x4_sRGB : MTLPixelFormatASTC_4x4_LDR;
5223 case QRhiTexture::ASTC_5x4:
5224 return srgb ? MTLPixelFormatASTC_5x4_sRGB : MTLPixelFormatASTC_5x4_LDR;
5225 case QRhiTexture::ASTC_5x5:
5226 return srgb ? MTLPixelFormatASTC_5x5_sRGB : MTLPixelFormatASTC_5x5_LDR;
5227 case QRhiTexture::ASTC_6x5:
5228 return srgb ? MTLPixelFormatASTC_6x5_sRGB : MTLPixelFormatASTC_6x5_LDR;
5229 case QRhiTexture::ASTC_6x6:
5230 return srgb ? MTLPixelFormatASTC_6x6_sRGB : MTLPixelFormatASTC_6x6_LDR;
5231 case QRhiTexture::ASTC_8x5:
5232 return srgb ? MTLPixelFormatASTC_8x5_sRGB : MTLPixelFormatASTC_8x5_LDR;
5233 case QRhiTexture::ASTC_8x6:
5234 return srgb ? MTLPixelFormatASTC_8x6_sRGB : MTLPixelFormatASTC_8x6_LDR;
5235 case QRhiTexture::ASTC_8x8:
5236 return srgb ? MTLPixelFormatASTC_8x8_sRGB : MTLPixelFormatASTC_8x8_LDR;
5237 case QRhiTexture::ASTC_10x5:
5238 return srgb ? MTLPixelFormatASTC_10x5_sRGB : MTLPixelFormatASTC_10x5_LDR;
5239 case QRhiTexture::ASTC_10x6:
5240 return srgb ? MTLPixelFormatASTC_10x6_sRGB : MTLPixelFormatASTC_10x6_LDR;
5241 case QRhiTexture::ASTC_10x8:
5242 return srgb ? MTLPixelFormatASTC_10x8_sRGB : MTLPixelFormatASTC_10x8_LDR;
5243 case QRhiTexture::ASTC_10x10:
5244 return srgb ? MTLPixelFormatASTC_10x10_sRGB : MTLPixelFormatASTC_10x10_LDR;
5245 case QRhiTexture::ASTC_12x10:
5246 return srgb ? MTLPixelFormatASTC_12x10_sRGB : MTLPixelFormatASTC_12x10_LDR;
5247 case QRhiTexture::ASTC_12x12:
5248 return srgb ? MTLPixelFormatASTC_12x12_sRGB : MTLPixelFormatASTC_12x12_LDR;
5249#else
5250 case QRhiTexture::ETC2_RGB8:
5251 if (d->caps.isAppleGPU)
5252 return srgb ? MTLPixelFormatETC2_RGB8_sRGB : MTLPixelFormatETC2_RGB8;
5253 qWarning("QRhiMetal: ETC2 compression not supported on this platform");
5254 return MTLPixelFormatInvalid;
5255 case QRhiTexture::ETC2_RGB8A1:
5256 if (d->caps.isAppleGPU)
5257 return srgb ? MTLPixelFormatETC2_RGB8A1_sRGB : MTLPixelFormatETC2_RGB8A1;
5258 qWarning("QRhiMetal: ETC2 compression not supported on this platform");
5259 return MTLPixelFormatInvalid;
5260 case QRhiTexture::ETC2_RGBA8:
5261 if (d->caps.isAppleGPU)
5262 return srgb ? MTLPixelFormatEAC_RGBA8_sRGB : MTLPixelFormatEAC_RGBA8;
5263 qWarning("QRhiMetal: ETC2 compression not supported on this platform");
5264 return MTLPixelFormatInvalid;
5265 case QRhiTexture::ASTC_4x4:
5266 if (d->caps.isAppleGPU)
5267 return srgb ? MTLPixelFormatASTC_4x4_sRGB : MTLPixelFormatASTC_4x4_LDR;
5268 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5269 return MTLPixelFormatInvalid;
5270 case QRhiTexture::ASTC_5x4:
5271 if (d->caps.isAppleGPU)
5272 return srgb ? MTLPixelFormatASTC_5x4_sRGB : MTLPixelFormatASTC_5x4_LDR;
5273 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5274 return MTLPixelFormatInvalid;
5275 case QRhiTexture::ASTC_5x5:
5276 if (d->caps.isAppleGPU)
5277 return srgb ? MTLPixelFormatASTC_5x5_sRGB : MTLPixelFormatASTC_5x5_LDR;
5278 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5279 return MTLPixelFormatInvalid;
5280 case QRhiTexture::ASTC_6x5:
5281 if (d->caps.isAppleGPU)
5282 return srgb ? MTLPixelFormatASTC_6x5_sRGB : MTLPixelFormatASTC_6x5_LDR;
5283 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5284 return MTLPixelFormatInvalid;
5285 case QRhiTexture::ASTC_6x6:
5286 if (d->caps.isAppleGPU)
5287 return srgb ? MTLPixelFormatASTC_6x6_sRGB : MTLPixelFormatASTC_6x6_LDR;
5288 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5289 return MTLPixelFormatInvalid;
5290 case QRhiTexture::ASTC_8x5:
5291 if (d->caps.isAppleGPU)
5292 return srgb ? MTLPixelFormatASTC_8x5_sRGB : MTLPixelFormatASTC_8x5_LDR;
5293 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5294 return MTLPixelFormatInvalid;
5295 case QRhiTexture::ASTC_8x6:
5296 if (d->caps.isAppleGPU)
5297 return srgb ? MTLPixelFormatASTC_8x6_sRGB : MTLPixelFormatASTC_8x6_LDR;
5298 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5299 return MTLPixelFormatInvalid;
5300 case QRhiTexture::ASTC_8x8:
5301 if (d->caps.isAppleGPU)
5302 return srgb ? MTLPixelFormatASTC_8x8_sRGB : MTLPixelFormatASTC_8x8_LDR;
5303 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5304 return MTLPixelFormatInvalid;
5305 case QRhiTexture::ASTC_10x5:
5306 if (d->caps.isAppleGPU)
5307 return srgb ? MTLPixelFormatASTC_10x5_sRGB : MTLPixelFormatASTC_10x5_LDR;
5308 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5309 return MTLPixelFormatInvalid;
5310 case QRhiTexture::ASTC_10x6:
5311 if (d->caps.isAppleGPU)
5312 return srgb ? MTLPixelFormatASTC_10x6_sRGB : MTLPixelFormatASTC_10x6_LDR;
5313 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5314 return MTLPixelFormatInvalid;
5315 case QRhiTexture::ASTC_10x8:
5316 if (d->caps.isAppleGPU)
5317 return srgb ? MTLPixelFormatASTC_10x8_sRGB : MTLPixelFormatASTC_10x8_LDR;
5318 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5319 return MTLPixelFormatInvalid;
5320 case QRhiTexture::ASTC_10x10:
5321 if (d->caps.isAppleGPU)
5322 return srgb ? MTLPixelFormatASTC_10x10_sRGB : MTLPixelFormatASTC_10x10_LDR;
5323 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5324 return MTLPixelFormatInvalid;
5325 case QRhiTexture::ASTC_12x10:
5326 if (d->caps.isAppleGPU)
5327 return srgb ? MTLPixelFormatASTC_12x10_sRGB : MTLPixelFormatASTC_12x10_LDR;
5328 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5329 return MTLPixelFormatInvalid;
5330 case QRhiTexture::ASTC_12x12:
5331 if (d->caps.isAppleGPU)
5332 return srgb ? MTLPixelFormatASTC_12x12_sRGB : MTLPixelFormatASTC_12x12_LDR;
5333 qWarning("QRhiMetal: ASTC compression not supported on this platform");
5334 return MTLPixelFormatInvalid;
5335#endif
5336
5337 default:
5338 Q_UNREACHABLE();
5339 return MTLPixelFormatInvalid;
5340 }
5341}
5342
5343QMetalRenderBuffer::QMetalRenderBuffer(QRhiImplementation *rhi, Type type, const QSize &pixelSize,
5344 int sampleCount, QRhiRenderBuffer::Flags flags,
5345 QRhiTexture::Format backingFormatHint)
5348{
5349}
5350
5352{
5353 destroy();
5354 delete d;
5355}
5356
5358{
5359 if (!d->tex)
5360 return;
5361
5365
5366 e.renderbuffer.texture = d->tex;
5367 d->tex = nil;
5368
5369 QRHI_RES_RHI(QRhiMetal);
5370 if (rhiD) {
5371 rhiD->d->releaseQueue.append(e);
5372 rhiD->unregisterResource(this);
5373 }
5374}
5375
5377{
5378 if (d->tex)
5379 destroy();
5380
5381 if (m_pixelSize.isEmpty())
5382 return false;
5383
5384 QRHI_RES_RHI(QRhiMetal);
5385 samples = rhiD->effectiveSampleCount(m_sampleCount);
5386
5387 MTLTextureDescriptor *desc = [[MTLTextureDescriptor alloc] init];
5388 desc.textureType = samples > 1 ? MTLTextureType2DMultisample : MTLTextureType2D;
5389 desc.width = NSUInteger(m_pixelSize.width());
5390 desc.height = NSUInteger(m_pixelSize.height());
5391 if (samples > 1)
5392 desc.sampleCount = NSUInteger(samples);
5393 desc.resourceOptions = MTLResourceStorageModePrivate;
5394 desc.usage = MTLTextureUsageRenderTarget;
5395
5396 // Memoryless contents cannot survive the render pass getting interrupted and
5397 // continued on another command encoder, which is what NoTransientBacking is
5398 // there to avoid.
5399 const bool canBeMemoryless = !m_flags.testFlag(QRhiRenderBuffer::NoTransientBacking);
5400
5401 switch (m_type) {
5402 case DepthStencil:
5403#ifdef Q_OS_MACOS
5404 if (rhiD->caps.isAppleGPU && canBeMemoryless) {
5405 desc.storageMode = MTLStorageModeMemoryless;
5406 d->format = MTLPixelFormatDepth32Float_Stencil8;
5407 } else {
5408 desc.storageMode = MTLStorageModePrivate;
5409 d->format = rhiD->d->dev.depth24Stencil8PixelFormatSupported
5410 ? MTLPixelFormatDepth24Unorm_Stencil8 : MTLPixelFormatDepth32Float_Stencil8;
5411 }
5412#else
5413 desc.storageMode = canBeMemoryless ? MTLStorageModeMemoryless : MTLStorageModePrivate;
5414 d->format = MTLPixelFormatDepth32Float_Stencil8;
5415#endif
5416 desc.pixelFormat = d->format;
5417 break;
5418 case Color:
5419 desc.storageMode = MTLStorageModePrivate;
5420 if (m_backingFormatHint != QRhiTexture::UnknownFormat)
5421 d->format = toMetalTextureFormat(m_backingFormatHint, {}, rhiD);
5422 else
5423 d->format = MTLPixelFormatRGBA8Unorm;
5424 desc.pixelFormat = d->format;
5425 break;
5426 default:
5427 Q_UNREACHABLE();
5428 break;
5429 }
5430
5431 d->tex = [rhiD->d->dev newTextureWithDescriptor: desc];
5432 [desc release];
5433
5434 if (!m_objectName.isEmpty())
5435 d->tex.label = [NSString stringWithUTF8String: m_objectName.constData()];
5436
5438 generation += 1;
5439 rhiD->registerResource(this);
5440 return true;
5441}
5442
5444{
5445 if (m_backingFormatHint != QRhiTexture::UnknownFormat)
5446 return m_backingFormatHint;
5447 else
5448 return m_type == Color ? QRhiTexture::RGBA8 : QRhiTexture::UnknownFormat;
5449}
5450
5451QMetalTexture::QMetalTexture(QRhiImplementation *rhi, Format format, const QSize &pixelSize, int depth,
5452 int arraySize, int sampleCount, Flags flags)
5454 d(new QMetalTextureData(this))
5455{
5456 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i)
5457 d->stagingBuf[i] = nil;
5458
5459 for (int i = 0; i < QRhi::MAX_MIP_LEVELS; ++i)
5460 d->perLevelViews[i] = nil;
5461}
5462
5464{
5465 destroy();
5466 delete d;
5467}
5468
5470{
5471 if (!d->tex)
5472 return;
5473
5477
5478 e.texture.texture = d->owns ? d->tex : nil;
5479 d->tex = nil;
5480
5481 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
5482 e.texture.stagingBuffers[i] = d->stagingBuf[i];
5483 d->stagingBuf[i] = nil;
5484 }
5485
5486 for (int i = 0; i < QRhi::MAX_MIP_LEVELS; ++i) {
5487 e.texture.views[i] = d->perLevelViews[i];
5488 d->perLevelViews[i] = nil;
5489 }
5490
5491 e.texture.samplingView = d->samplingView;
5492 d->samplingView = nil;
5493 e.texture.writeView = d->writeView;
5494 d->writeView = nil;
5495
5496 QRHI_RES_RHI(QRhiMetal);
5497 if (rhiD) {
5498 rhiD->d->releaseQueue.append(e);
5499 rhiD->unregisterResource(this);
5500 }
5501}
5502
5503bool QMetalTexture::prepareCreate(QSize *adjustedSize)
5504{
5505 if (d->tex)
5506 destroy();
5507
5508 const bool isCube = m_flags.testFlag(CubeMap);
5509 const bool is3D = m_flags.testFlag(ThreeDimensional);
5510 const bool isArray = m_flags.testFlag(TextureArray);
5511 const bool hasMipMaps = m_flags.testFlag(MipMapped);
5512 const bool is1D = m_flags.testFlag(OneDimensional);
5513
5514 const QSize size = is1D ? QSize(qMax(1, m_pixelSize.width()), 1)
5515 : (m_pixelSize.isEmpty() ? QSize(1, 1) : m_pixelSize);
5516
5517 QRHI_RES_RHI(QRhiMetal);
5518 d->format = toMetalTextureFormat(m_format, m_flags, rhiD);
5519 if (m_writeViewFormat.format != UnknownFormat) {
5520 d->viewFormat = toMetalTextureFormat(m_writeViewFormat.format,
5521 m_writeViewFormat.srgb ? sRGB : Flags(), rhiD);
5522 } else {
5523 d->viewFormat = d->format;
5524 }
5525 if (m_readViewFormat.format != UnknownFormat) {
5526 d->viewFormatForSampling = toMetalTextureFormat(m_readViewFormat.format,
5527 m_readViewFormat.srgb ? sRGB : Flags(), rhiD);
5528 } else {
5529 d->viewFormatForSampling = d->format;
5530 }
5531 mipLevelCount = hasMipMaps ? rhiD->q->mipLevelsForSize(size) : 1;
5532 samples = rhiD->effectiveSampleCount(m_sampleCount);
5533 if (samples > 1) {
5534 if (isCube) {
5535 qWarning("Cubemap texture cannot be multisample");
5536 return false;
5537 }
5538 if (is3D) {
5539 qWarning("3D texture cannot be multisample");
5540 return false;
5541 }
5542 if (hasMipMaps) {
5543 qWarning("Multisample texture cannot have mipmaps");
5544 return false;
5545 }
5546 }
5547 if (isCube && is3D) {
5548 qWarning("Texture cannot be both cube and 3D");
5549 return false;
5550 }
5551 if (isArray && is3D) {
5552 qWarning("Texture cannot be both array and 3D");
5553 return false;
5554 }
5555 if (is1D && is3D) {
5556 qWarning("Texture cannot be both 1D and 3D");
5557 return false;
5558 }
5559 if (is1D && isCube) {
5560 qWarning("Texture cannot be both 1D and cube");
5561 return false;
5562 }
5563 if (m_depth > 1 && !is3D) {
5564 qWarning("Texture cannot have a depth of %d when it is not 3D", m_depth);
5565 return false;
5566 }
5567 if (m_arraySize > 0 && !isArray) {
5568 qWarning("Texture cannot have an array size of %d when it is not an array", m_arraySize);
5569 return false;
5570 }
5571 if (m_arraySize < 1 && isArray) {
5572 qWarning("Texture is an array but array size is %d", m_arraySize);
5573 return false;
5574 }
5575
5576 if (!rhiD->textureFormatInfo(m_format, size, nullptr, nullptr, nullptr))
5577 return false;
5578
5579 if (adjustedSize)
5580 *adjustedSize = size;
5581
5582 return true;
5583}
5584
5586{
5587 QSize size;
5588 if (!prepareCreate(&size))
5589 return false;
5590
5591 MTLTextureDescriptor *desc = [[MTLTextureDescriptor alloc] init];
5592
5593 const bool isCube = m_flags.testFlag(CubeMap);
5594 const bool is3D = m_flags.testFlag(ThreeDimensional);
5595 const bool isArray = m_flags.testFlag(TextureArray);
5596 const bool is1D = m_flags.testFlag(OneDimensional);
5597 if (isCube) {
5598 desc.textureType = MTLTextureTypeCube;
5599 } else if (is3D) {
5600 desc.textureType = MTLTextureType3D;
5601 } else if (is1D) {
5602 desc.textureType = isArray ? MTLTextureType1DArray : MTLTextureType1D;
5603 } else if (isArray) {
5604 desc.textureType = samples > 1 ? MTLTextureType2DMultisampleArray : MTLTextureType2DArray;
5605 } else {
5606 desc.textureType = samples > 1 ? MTLTextureType2DMultisample : MTLTextureType2D;
5607 }
5608 desc.pixelFormat = d->format;
5609 desc.width = NSUInteger(size.width());
5610 desc.height = NSUInteger(size.height());
5611 desc.depth = is3D ? qMax(1, m_depth) : 1;
5612 desc.mipmapLevelCount = NSUInteger(mipLevelCount);
5613 if (samples > 1)
5614 desc.sampleCount = NSUInteger(samples);
5615 if (isArray)
5616 desc.arrayLength = NSUInteger(qMax(0, m_arraySize));
5617 desc.resourceOptions = MTLResourceStorageModePrivate;
5618 desc.storageMode = MTLStorageModePrivate;
5619 desc.usage = MTLTextureUsageShaderRead;
5620 if (m_flags.testFlag(RenderTarget))
5621 desc.usage |= MTLTextureUsageRenderTarget;
5622 if (m_flags.testFlag(UsedWithLoadStore))
5623 desc.usage |= MTLTextureUsageShaderWrite;
5624
5625 // Metal requires this only when the view changes the component layout, and
5626 // explicitly says not to set it for linear <-> sRGB views. A view of the
5627 // same QRhi format with a different sRGB setting, or over an array slice
5628 // range, therefore does not need it.
5629 const bool writeViewChangesFormat = m_writeViewFormat.format != UnknownFormat
5630 && m_writeViewFormat.format != m_format;
5631 const bool readViewChangesFormat = m_readViewFormat.format != UnknownFormat
5632 && m_readViewFormat.format != m_format;
5633 if (writeViewChangesFormat || readViewChangesFormat)
5634 desc.usage |= MTLTextureUsagePixelFormatView;
5635
5636 QRHI_RES_RHI(QRhiMetal);
5637 d->tex = [rhiD->d->dev newTextureWithDescriptor: desc];
5638 [desc release];
5639
5640 if (!m_objectName.isEmpty())
5641 d->tex.label = [NSString stringWithUTF8String: m_objectName.constData()];
5642
5643 d->owns = true;
5644
5645 if (!d->createViews())
5646 return false;
5647
5649 generation += 1;
5650 rhiD->registerResource(this);
5651 return true;
5652}
5653
5654bool QMetalTexture::createFrom(QRhiTexture::NativeTexture src)
5655{
5656 id<MTLTexture> tex = id<MTLTexture>(src.object);
5657 if (tex == 0)
5658 return false;
5659
5660 if (!prepareCreate())
5661 return false;
5662
5663 d->tex = tex;
5664
5665 d->owns = false;
5666
5667 if (!d->createViews())
5668 return false;
5669
5671 generation += 1;
5672 QRHI_RES_RHI(QRhiMetal);
5673 rhiD->registerResource(this);
5674 return true;
5675}
5676
5678{
5679 return {quint64(d->tex), 0};
5680}
5681
5683{
5684 Q_ASSERT(!samplingView && !writeView);
5685
5686 const bool isCube = q->m_flags.testFlag(QRhiTexture::CubeMap);
5687 const bool isArray = q->m_flags.testFlag(QRhiTexture::TextureArray);
5688 const bool hasArrayRange = isArray && q->m_arrayRangeStart >= 0 && q->m_arrayRangeLength >= 0;
5689 const NSUInteger sliceCount = isCube ? 6 : (isArray ? NSUInteger(qMax(0, q->m_arraySize)) : 1);
5690 const NSRange levels = NSMakeRange(0, NSUInteger(q->mipLevelCount));
5691 const MTLTextureType type = [tex textureType];
5692
5693 id<MTLTexture> newSamplingView = nil;
5694 id<MTLTexture> newWriteView = nil;
5695
5696 if (viewFormatForSampling != format || hasArrayRange) {
5697 const NSRange slices = hasArrayRange
5698 ? NSMakeRange(NSUInteger(q->m_arrayRangeStart), NSUInteger(q->m_arrayRangeLength))
5699 : NSMakeRange(0, sliceCount);
5700 newSamplingView = [tex newTextureViewWithPixelFormat: viewFormatForSampling
5701 textureType: type levels: levels slices: slices];
5702 if (!newSamplingView) {
5703 qWarning("QRhiMetal: Failed to create texture view used for sampling");
5704 return false;
5705 }
5706 }
5707
5708 if (viewFormat != format) {
5709 newWriteView = [tex newTextureViewWithPixelFormat: viewFormat
5710 textureType: type levels: levels
5711 slices: NSMakeRange(0, sliceCount)];
5712 if (!newWriteView) {
5713 qWarning("QRhiMetal: Failed to create texture view used for rendering");
5714 [newSamplingView release];
5715 return false;
5716 }
5717 }
5718
5719 samplingView = newSamplingView;
5720 writeView = newWriteView;
5721 return true;
5722}
5723
5725{
5726 Q_ASSERT(level >= 0 && level < int(q->mipLevelCount));
5727 if (perLevelViews[level])
5728 return perLevelViews[level];
5729
5730 const MTLTextureType type = [tex textureType];
5731 const bool isCube = q->m_flags.testFlag(QRhiTexture::CubeMap);
5732 const bool isArray = q->m_flags.testFlag(QRhiTexture::TextureArray);
5733 id<MTLTexture> view = [tex newTextureViewWithPixelFormat: viewFormat textureType: type
5734 levels: NSMakeRange(NSUInteger(level), 1)
5735 slices: NSMakeRange(0, isCube ? 6 : (isArray ? qMax(0, q->m_arraySize) : 1))];
5736
5737 perLevelViews[level] = view;
5738 return view;
5739}
5740
5741QMetalSampler::QMetalSampler(QRhiImplementation *rhi, Filter magFilter, Filter minFilter, Filter mipmapMode,
5742 AddressMode u, AddressMode v, AddressMode w)
5744 d(new QMetalSamplerData)
5745{
5746}
5747
5749{
5750 destroy();
5751 delete d;
5752}
5753
5755{
5756 if (!d->samplerState)
5757 return;
5758
5762
5763 e.sampler.samplerState = d->samplerState;
5764 d->samplerState = nil;
5765
5766 QRHI_RES_RHI(QRhiMetal);
5767 if (rhiD) {
5768 rhiD->d->releaseQueue.append(e);
5769 rhiD->unregisterResource(this);
5770 }
5771}
5772
5773static inline MTLSamplerMinMagFilter toMetalFilter(QRhiSampler::Filter f)
5774{
5775 switch (f) {
5776 case QRhiSampler::Nearest:
5777 return MTLSamplerMinMagFilterNearest;
5778 case QRhiSampler::Linear:
5779 return MTLSamplerMinMagFilterLinear;
5780 default:
5781 Q_UNREACHABLE();
5782 return MTLSamplerMinMagFilterNearest;
5783 }
5784}
5785
5786static inline MTLSamplerMipFilter toMetalMipmapMode(QRhiSampler::Filter f)
5787{
5788 switch (f) {
5789 case QRhiSampler::None:
5790 return MTLSamplerMipFilterNotMipmapped;
5791 case QRhiSampler::Nearest:
5792 return MTLSamplerMipFilterNearest;
5793 case QRhiSampler::Linear:
5794 return MTLSamplerMipFilterLinear;
5795 default:
5796 Q_UNREACHABLE();
5797 return MTLSamplerMipFilterNotMipmapped;
5798 }
5799}
5800
5801static inline MTLSamplerAddressMode toMetalAddressMode(QRhiSampler::AddressMode m)
5802{
5803 switch (m) {
5804 case QRhiSampler::Repeat:
5805 return MTLSamplerAddressModeRepeat;
5806 case QRhiSampler::ClampToEdge:
5807 return MTLSamplerAddressModeClampToEdge;
5808 case QRhiSampler::Mirror:
5809 return MTLSamplerAddressModeMirrorRepeat;
5810 default:
5811 Q_UNREACHABLE();
5812 return MTLSamplerAddressModeClampToEdge;
5813 }
5814}
5815
5816static inline MTLCompareFunction toMetalTextureCompareFunction(QRhiSampler::CompareOp op)
5817{
5818 switch (op) {
5819 case QRhiSampler::Never:
5820 return MTLCompareFunctionNever;
5821 case QRhiSampler::Less:
5822 return MTLCompareFunctionLess;
5823 case QRhiSampler::Equal:
5824 return MTLCompareFunctionEqual;
5825 case QRhiSampler::LessOrEqual:
5826 return MTLCompareFunctionLessEqual;
5827 case QRhiSampler::Greater:
5828 return MTLCompareFunctionGreater;
5829 case QRhiSampler::NotEqual:
5830 return MTLCompareFunctionNotEqual;
5831 case QRhiSampler::GreaterOrEqual:
5832 return MTLCompareFunctionGreaterEqual;
5833 case QRhiSampler::Always:
5834 return MTLCompareFunctionAlways;
5835 default:
5836 Q_UNREACHABLE();
5837 return MTLCompareFunctionNever;
5838 }
5839}
5840
5842{
5843 if (d->samplerState)
5844 destroy();
5845
5846 MTLSamplerDescriptor *desc = [[MTLSamplerDescriptor alloc] init];
5847 desc.minFilter = toMetalFilter(m_minFilter);
5848 desc.magFilter = toMetalFilter(m_magFilter);
5849 desc.mipFilter = toMetalMipmapMode(m_mipmapMode);
5850 desc.sAddressMode = toMetalAddressMode(m_addressU);
5851 desc.tAddressMode = toMetalAddressMode(m_addressV);
5852 desc.rAddressMode = toMetalAddressMode(m_addressW);
5853 desc.compareFunction = toMetalTextureCompareFunction(m_compareOp);
5854 QRHI_RES_RHI(QRhiMetal);
5855 // Whether this sampler ends up in an argument buffer is not known here, and
5856 // a sampler state cannot be changed afterwards. Metal gives no warning when
5857 // one without this is used with an argument buffer, it just faults the GPU.
5858 // So set it always as long as ICBs are supported.
5859 desc.supportArgumentBuffers = rhiD->caps.indirectCommandBuffers ? YES : NO;
5860 d->samplerState = [rhiD->d->dev newSamplerStateWithDescriptor: desc];
5861 [desc release];
5862 if (!d->samplerState) {
5863 // Sampler states with supportArgumentBuffers set draw on a per-process
5864 // quota (MTLDevice.maxArgumentBufferSamplerCount), so mention that as
5865 // the likely reason for a failure that is otherwise hard to explain.
5866 qWarning("Failed to create Metal sampler state. The number of unique sampler "
5867 "states with argument buffer support may have exceeded the limit of %u.",
5868 uint(rhiD->d->dev.maxArgumentBufferSamplerCount));
5869 return false;
5870 }
5871
5873 generation += 1;
5874 rhiD->registerResource(this);
5875 return true;
5876}
5877
5881{
5882}
5883
5885{
5886 destroy();
5887 delete d;
5888}
5889
5891{
5892 if (!d->rateMap)
5893 return;
5894
5898
5899 e.shadingRateMap.rateMap = d->rateMap;
5900 d->rateMap = nil;
5901
5902 QRHI_RES_RHI(QRhiMetal);
5903 if (rhiD) {
5904 rhiD->d->releaseQueue.append(e);
5905 rhiD->unregisterResource(this);
5906 }
5907}
5908
5909bool QMetalShadingRateMap::createFrom(NativeShadingRateMap src)
5910{
5911 if (d->rateMap)
5912 destroy();
5913
5914 d->rateMap = (id<MTLRasterizationRateMap>) (quintptr(src.object));
5915 if (!d->rateMap)
5916 return false;
5917
5918 [d->rateMap retain];
5919
5921 generation += 1;
5922 QRHI_RES_RHI(QRhiMetal);
5923 rhiD->registerResource(this);
5924 return true;
5925}
5926
5928{
5929 if (!d->rateMap)
5930 return QSize();
5931
5932 const MTLSize screenSize = [d->rateMap screenSize];
5933 return QSize(int(screenSize.width), int(screenSize.height));
5934}
5935
5936// dummy, no Vulkan-style RenderPass+Framebuffer concept here.
5937// We do have MTLRenderPassDescriptor of course, but it will be created on the fly for each pass.
5940{
5941 serializedFormatData.reserve(16);
5942}
5943
5948
5950{
5951 QRHI_RES_RHI(QRhiMetal);
5952 if (rhiD)
5953 rhiD->unregisterResource(this);
5954}
5955
5956bool QMetalRenderPassDescriptor::isCompatible(const QRhiRenderPassDescriptor *other) const
5957{
5958 if (!other)
5959 return false;
5960
5962
5964 return false;
5965
5967 return false;
5968
5969 for (int i = 0; i < colorAttachmentCount; ++i) {
5970 if (colorFormat[i] != o->colorFormat[i])
5971 return false;
5972 }
5973
5974 if (hasDepthStencil) {
5975 if (dsFormat != o->dsFormat)
5976 return false;
5977 }
5978
5980 return false;
5981
5982 return true;
5983}
5984
5986{
5987 serializedFormatData.clear();
5988 auto p = std::back_inserter(serializedFormatData);
5989
5990 *p++ = colorAttachmentCount;
5991 *p++ = hasDepthStencil;
5992 for (int i = 0; i < colorAttachmentCount; ++i)
5993 *p++ = colorFormat[i];
5994 *p++ = hasDepthStencil ? dsFormat : 0;
5995 *p++ = hasShadingRateMap;
5996}
5997
5999{
6000 QMetalRenderPassDescriptor *rpD = new QMetalRenderPassDescriptor(m_rhi);
6003 memcpy(rpD->colorFormat, colorFormat, sizeof(colorFormat));
6004 rpD->dsFormat = dsFormat;
6006
6008
6009 QRHI_RES_RHI(QRhiMetal);
6010 rhiD->registerResource(rpD, false);
6011 return rpD;
6012}
6013
6015{
6016 return serializedFormatData;
6017}
6018
6019QMetalSwapChainRenderTarget::QMetalSwapChainRenderTarget(QRhiImplementation *rhi, QRhiSwapChain *swapchain)
6022{
6023}
6024
6030
6032{
6033 // nothing to do here
6034}
6035
6037{
6038 return d->pixelSize;
6039}
6040
6042{
6043 return d->dpr;
6044}
6045
6047{
6048 return d->sampleCount;
6049}
6050
6052 const QRhiTextureRenderTargetDescription &desc,
6053 Flags flags)
6056{
6057}
6058
6064
6066{
6067 QRHI_RES_RHI(QRhiMetal);
6068 if (rhiD)
6069 rhiD->unregisterResource(this);
6070}
6071
6073{
6074 const int colorAttachmentCount = int(m_desc.colorAttachmentCount());
6075 QMetalRenderPassDescriptor *rpD = new QMetalRenderPassDescriptor(m_rhi);
6076 rpD->colorAttachmentCount = colorAttachmentCount;
6077 rpD->hasDepthStencil = m_desc.depthStencilBuffer() || m_desc.depthTexture();
6078
6079 for (int i = 0; i < colorAttachmentCount; ++i) {
6080 const QRhiColorAttachment *colorAtt = m_desc.colorAttachmentAt(i);
6081 QMetalTexture *texD = QRHI_RES(QMetalTexture, colorAtt->texture());
6082 QMetalRenderBuffer *rbD = QRHI_RES(QMetalRenderBuffer, colorAtt->renderBuffer());
6083 rpD->colorFormat[i] = int(texD ? texD->d->viewFormat : rbD->d->format);
6084 }
6085
6086 if (m_desc.depthTexture())
6087 rpD->dsFormat = int(QRHI_RES(QMetalTexture, m_desc.depthTexture())->d->viewFormat);
6088 else if (m_desc.depthStencilBuffer())
6089 rpD->dsFormat = int(QRHI_RES(QMetalRenderBuffer, m_desc.depthStencilBuffer())->d->format);
6090
6091 rpD->hasShadingRateMap = m_desc.shadingRateMap() != nullptr;
6092
6094
6095 QRHI_RES_RHI(QRhiMetal);
6096 rhiD->registerResource(rpD, false);
6097 return rpD;
6098}
6099
6101{
6102 QRHI_RES_RHI(QRhiMetal);
6103 Q_ASSERT(m_desc.colorAttachmentCount() > 0 || m_desc.depthTexture());
6104 Q_ASSERT(!m_desc.depthStencilBuffer() || !m_desc.depthTexture());
6105 const bool hasDepthStencil = m_desc.depthStencilBuffer() || m_desc.depthTexture();
6106
6107 d->colorAttCount = 0;
6108 int attIndex = 0;
6109 for (auto it = m_desc.cbeginColorAttachments(), itEnd = m_desc.cendColorAttachments(); it != itEnd; ++it, ++attIndex) {
6110 d->colorAttCount += 1;
6111 QMetalTexture *texD = QRHI_RES(QMetalTexture, it->texture());
6112 QMetalRenderBuffer *rbD = QRHI_RES(QMetalRenderBuffer, it->renderBuffer());
6113 Q_ASSERT(texD || rbD);
6114 id<MTLTexture> dst = nil;
6115 bool is3D = false;
6116 if (texD) {
6117 dst = texD->d->textureForWrite();
6118 if (attIndex == 0) {
6119 d->pixelSize = rhiD->q->sizeForMipLevel(it->level(), texD->pixelSize());
6121 }
6122 is3D = texD->flags().testFlag(QRhiTexture::ThreeDimensional);
6123 } else if (rbD) {
6124 dst = rbD->d->tex;
6125 if (attIndex == 0) {
6126 d->pixelSize = rbD->pixelSize();
6128 }
6129 }
6131 colorAtt.tex = dst;
6132 colorAtt.arrayLayer = is3D ? 0 : it->layer();
6133 colorAtt.slice = is3D ? it->layer() : 0;
6134 colorAtt.level = it->level();
6135 QMetalTexture *resTexD = QRHI_RES(QMetalTexture, it->resolveTexture());
6136 colorAtt.resolveTex = resTexD ? resTexD->d->textureForWrite() : nil;
6137 colorAtt.resolveLayer = it->resolveLayer();
6138 colorAtt.resolveLevel = it->resolveLevel();
6139 d->fb.colorAtt[attIndex] = colorAtt;
6140 }
6141 d->dpr = 1;
6142
6143 if (hasDepthStencil) {
6144 if (m_desc.depthTexture()) {
6145 QMetalTexture *depthTexD = QRHI_RES(QMetalTexture, m_desc.depthTexture());
6146 d->fb.dsTex = depthTexD->d->textureForWrite();
6147 d->fb.hasStencil = rhiD->isStencilSupportingFormat(depthTexD->format());
6148 d->fb.depthNeedsStore = !m_flags.testFlag(DoNotStoreDepthStencilContents) && !m_desc.depthResolveTexture();
6149 d->fb.preserveDs = m_flags.testFlag(QRhiTextureRenderTarget::PreserveDepthStencilContents);
6150 if (d->colorAttCount == 0) {
6151 d->pixelSize = depthTexD->pixelSize();
6152 d->sampleCount = depthTexD->samples;
6153 }
6154 } else {
6155 QMetalRenderBuffer *depthRbD = QRHI_RES(QMetalRenderBuffer, m_desc.depthStencilBuffer());
6156 d->fb.dsTex = depthRbD->d->tex;
6157 d->fb.hasStencil = true;
6158 d->fb.depthNeedsStore = false;
6159 d->fb.preserveDs = false;
6160 if (d->colorAttCount == 0) {
6161 d->pixelSize = depthRbD->pixelSize();
6162 d->sampleCount = depthRbD->samples;
6163 }
6164 }
6165 if (m_desc.depthResolveTexture()) {
6166 QMetalTexture *depthResolveTexD = QRHI_RES(QMetalTexture, m_desc.depthResolveTexture());
6167 d->fb.dsResolveTex = depthResolveTexD->d->textureForWrite();
6168 }
6169 d->dsAttCount = 1;
6170 } else {
6171 d->dsAttCount = 0;
6172 }
6173
6174 if (d->colorAttCount > 0)
6175 d->fb.preserveColor = m_flags.testFlag(QRhiTextureRenderTarget::PreserveColorContents);
6176
6177 QRhiRenderTargetAttachmentTracker::updateResIdList<QMetalTexture, QMetalRenderBuffer>(m_desc, &d->currentResIdList);
6178
6179 rhiD->registerResource(this, false);
6180 return true;
6181}
6182
6184{
6185 if (!QRhiRenderTargetAttachmentTracker::isUpToDate<QMetalTexture, QMetalRenderBuffer>(m_desc, d->currentResIdList))
6186 const_cast<QMetalTextureRenderTarget *>(this)->create();
6187
6188 return d->pixelSize;
6189}
6190
6192{
6193 return d->dpr;
6194}
6195
6197{
6198 return d->sampleCount;
6199}
6200
6205
6210
6212{
6213 sortedBindings.clear();
6214 maxBinding = -1;
6215
6216 QRHI_RES_RHI(QRhiMetal);
6217 if (rhiD)
6218 rhiD->unregisterResource(this);
6219}
6220
6222{
6223 if (!sortedBindings.isEmpty())
6224 destroy();
6225
6226 QRHI_RES_RHI(QRhiMetal);
6227 if (!rhiD->sanityCheckShaderResourceBindings(this))
6228 return false;
6229
6230 rhiD->updateLayoutDesc(this);
6231
6232 std::copy(m_bindings.cbegin(), m_bindings.cend(), std::back_inserter(sortedBindings));
6233 std::sort(sortedBindings.begin(), sortedBindings.end(), QRhiImplementation::sortedBindingLessThan);
6234 if (!sortedBindings.isEmpty())
6235 maxBinding = QRhiImplementation::shaderResourceBindingData(sortedBindings.last())->binding;
6236 else
6237 maxBinding = -1;
6238
6239 boundResourceData.resize(sortedBindings.count());
6240
6241 for (BoundResourceData &bd : boundResourceData)
6242 bd = {};
6243
6244 generation += 1;
6245 rhiD->registerResource(this, false);
6246 return true;
6247}
6248
6250{
6251 sortedBindings.clear();
6252 std::copy(m_bindings.cbegin(), m_bindings.cend(), std::back_inserter(sortedBindings));
6253 if (!flags.testFlag(BindingsAreSorted))
6254 std::sort(sortedBindings.begin(), sortedBindings.end(), QRhiImplementation::sortedBindingLessThan);
6255
6256 for (BoundResourceData &bd : boundResourceData)
6257 bd = {};
6258
6259 generation += 1;
6260}
6261
6265{
6266 d->q = this;
6267 d->tess.q = d;
6268}
6269
6275
6277{
6278 d->vs.destroy();
6279 d->fs.destroy();
6280
6281 d->icbCapable = false;
6282
6283 d->tess.compVs[0].destroy();
6284 d->tess.compVs[1].destroy();
6285 d->tess.compVs[2].destroy();
6286
6287 d->tess.compTesc.destroy();
6288 d->tess.vertTese.destroy();
6289
6290 qDeleteAll(d->extraBufMgr.deviceLocalWorkBuffers);
6291 d->extraBufMgr.deviceLocalWorkBuffers.clear();
6292 qDeleteAll(d->extraBufMgr.hostVisibleWorkBuffers);
6293 d->extraBufMgr.hostVisibleWorkBuffers.clear();
6294
6295 delete d->bufferSizeBuffer;
6296 d->bufferSizeBuffer = nullptr;
6297
6298 if (!d->ps && !d->ds
6299 && !d->tess.vertexComputeState[0] && !d->tess.vertexComputeState[1] && !d->tess.vertexComputeState[2]
6300 && !d->tess.tessControlComputeState)
6301 {
6302 return;
6303 }
6304
6308 e.graphicsPipeline.pipelineState = d->ps;
6309 e.graphicsPipeline.depthStencilState = d->ds;
6310 e.graphicsPipeline.tessVertexComputeState = d->tess.vertexComputeState;
6311 e.graphicsPipeline.tessTessControlComputeState = d->tess.tessControlComputeState;
6312 d->ps = nil;
6313 d->ds = nil;
6314 d->tess.vertexComputeState = {};
6315 d->tess.tessControlComputeState = nil;
6316
6317 QRHI_RES_RHI(QRhiMetal);
6318 if (rhiD) {
6319 rhiD->d->releaseQueue.append(e);
6320 rhiD->unregisterResource(this);
6321 }
6322}
6323
6324static inline MTLVertexFormat toMetalAttributeFormat(QRhiVertexInputAttribute::Format format)
6325{
6326 switch (format) {
6327 case QRhiVertexInputAttribute::Float4:
6328 return MTLVertexFormatFloat4;
6329 case QRhiVertexInputAttribute::Float3:
6330 return MTLVertexFormatFloat3;
6331 case QRhiVertexInputAttribute::Float2:
6332 return MTLVertexFormatFloat2;
6333 case QRhiVertexInputAttribute::Float:
6334 return MTLVertexFormatFloat;
6335 case QRhiVertexInputAttribute::UNormByte4:
6336 return MTLVertexFormatUChar4Normalized;
6337 case QRhiVertexInputAttribute::UNormByte2:
6338 return MTLVertexFormatUChar2Normalized;
6339 case QRhiVertexInputAttribute::UNormByte:
6340 return MTLVertexFormatUCharNormalized;
6341 case QRhiVertexInputAttribute::UInt4:
6342 return MTLVertexFormatUInt4;
6343 case QRhiVertexInputAttribute::UInt3:
6344 return MTLVertexFormatUInt3;
6345 case QRhiVertexInputAttribute::UInt2:
6346 return MTLVertexFormatUInt2;
6347 case QRhiVertexInputAttribute::UInt:
6348 return MTLVertexFormatUInt;
6349 case QRhiVertexInputAttribute::SInt4:
6350 return MTLVertexFormatInt4;
6351 case QRhiVertexInputAttribute::SInt3:
6352 return MTLVertexFormatInt3;
6353 case QRhiVertexInputAttribute::SInt2:
6354 return MTLVertexFormatInt2;
6355 case QRhiVertexInputAttribute::SInt:
6356 return MTLVertexFormatInt;
6357 case QRhiVertexInputAttribute::Half4:
6358 return MTLVertexFormatHalf4;
6359 case QRhiVertexInputAttribute::Half3:
6360 return MTLVertexFormatHalf3;
6361 case QRhiVertexInputAttribute::Half2:
6362 return MTLVertexFormatHalf2;
6363 case QRhiVertexInputAttribute::Half:
6364 return MTLVertexFormatHalf;
6365 case QRhiVertexInputAttribute::UShort4:
6366 return MTLVertexFormatUShort4;
6367 case QRhiVertexInputAttribute::UShort3:
6368 return MTLVertexFormatUShort3;
6369 case QRhiVertexInputAttribute::UShort2:
6370 return MTLVertexFormatUShort2;
6371 case QRhiVertexInputAttribute::UShort:
6372 return MTLVertexFormatUShort;
6373 case QRhiVertexInputAttribute::SShort4:
6374 return MTLVertexFormatShort4;
6375 case QRhiVertexInputAttribute::SShort3:
6376 return MTLVertexFormatShort3;
6377 case QRhiVertexInputAttribute::SShort2:
6378 return MTLVertexFormatShort2;
6379 case QRhiVertexInputAttribute::SShort:
6380 return MTLVertexFormatShort;
6381 default:
6382 Q_UNREACHABLE();
6383 return MTLVertexFormatFloat4;
6384 }
6385}
6386
6387static inline MTLBlendFactor toMetalBlendFactor(QRhiGraphicsPipeline::BlendFactor f)
6388{
6389 switch (f) {
6390 case QRhiGraphicsPipeline::Zero:
6391 return MTLBlendFactorZero;
6392 case QRhiGraphicsPipeline::One:
6393 return MTLBlendFactorOne;
6394 case QRhiGraphicsPipeline::SrcColor:
6395 return MTLBlendFactorSourceColor;
6396 case QRhiGraphicsPipeline::OneMinusSrcColor:
6397 return MTLBlendFactorOneMinusSourceColor;
6398 case QRhiGraphicsPipeline::DstColor:
6399 return MTLBlendFactorDestinationColor;
6400 case QRhiGraphicsPipeline::OneMinusDstColor:
6401 return MTLBlendFactorOneMinusDestinationColor;
6402 case QRhiGraphicsPipeline::SrcAlpha:
6403 return MTLBlendFactorSourceAlpha;
6404 case QRhiGraphicsPipeline::OneMinusSrcAlpha:
6405 return MTLBlendFactorOneMinusSourceAlpha;
6406 case QRhiGraphicsPipeline::DstAlpha:
6407 return MTLBlendFactorDestinationAlpha;
6408 case QRhiGraphicsPipeline::OneMinusDstAlpha:
6409 return MTLBlendFactorOneMinusDestinationAlpha;
6410 case QRhiGraphicsPipeline::ConstantColor:
6411 return MTLBlendFactorBlendColor;
6412 case QRhiGraphicsPipeline::ConstantAlpha:
6413 return MTLBlendFactorBlendAlpha;
6414 case QRhiGraphicsPipeline::OneMinusConstantColor:
6415 return MTLBlendFactorOneMinusBlendColor;
6416 case QRhiGraphicsPipeline::OneMinusConstantAlpha:
6417 return MTLBlendFactorOneMinusBlendAlpha;
6418 case QRhiGraphicsPipeline::SrcAlphaSaturate:
6419 return MTLBlendFactorSourceAlphaSaturated;
6420 case QRhiGraphicsPipeline::Src1Color:
6421 return MTLBlendFactorSource1Color;
6422 case QRhiGraphicsPipeline::OneMinusSrc1Color:
6423 return MTLBlendFactorOneMinusSource1Color;
6424 case QRhiGraphicsPipeline::Src1Alpha:
6425 return MTLBlendFactorSource1Alpha;
6426 case QRhiGraphicsPipeline::OneMinusSrc1Alpha:
6427 return MTLBlendFactorOneMinusSource1Alpha;
6428 default:
6429 Q_UNREACHABLE();
6430 return MTLBlendFactorZero;
6431 }
6432}
6433
6434static inline MTLBlendOperation toMetalBlendOp(QRhiGraphicsPipeline::BlendOp op)
6435{
6436 switch (op) {
6437 case QRhiGraphicsPipeline::Add:
6438 return MTLBlendOperationAdd;
6439 case QRhiGraphicsPipeline::Subtract:
6440 return MTLBlendOperationSubtract;
6441 case QRhiGraphicsPipeline::ReverseSubtract:
6442 return MTLBlendOperationReverseSubtract;
6443 case QRhiGraphicsPipeline::Min:
6444 return MTLBlendOperationMin;
6445 case QRhiGraphicsPipeline::Max:
6446 return MTLBlendOperationMax;
6447 default:
6448 Q_UNREACHABLE();
6449 return MTLBlendOperationAdd;
6450 }
6451}
6452
6453static inline uint toMetalColorWriteMask(QRhiGraphicsPipeline::ColorMask c)
6454{
6455 uint f = 0;
6456 if (c.testFlag(QRhiGraphicsPipeline::R))
6457 f |= MTLColorWriteMaskRed;
6458 if (c.testFlag(QRhiGraphicsPipeline::G))
6459 f |= MTLColorWriteMaskGreen;
6460 if (c.testFlag(QRhiGraphicsPipeline::B))
6461 f |= MTLColorWriteMaskBlue;
6462 if (c.testFlag(QRhiGraphicsPipeline::A))
6463 f |= MTLColorWriteMaskAlpha;
6464 return f;
6465}
6466
6467static inline MTLCompareFunction toMetalCompareOp(QRhiGraphicsPipeline::CompareOp op)
6468{
6469 switch (op) {
6470 case QRhiGraphicsPipeline::Never:
6471 return MTLCompareFunctionNever;
6472 case QRhiGraphicsPipeline::Less:
6473 return MTLCompareFunctionLess;
6474 case QRhiGraphicsPipeline::Equal:
6475 return MTLCompareFunctionEqual;
6476 case QRhiGraphicsPipeline::LessOrEqual:
6477 return MTLCompareFunctionLessEqual;
6478 case QRhiGraphicsPipeline::Greater:
6479 return MTLCompareFunctionGreater;
6480 case QRhiGraphicsPipeline::NotEqual:
6481 return MTLCompareFunctionNotEqual;
6482 case QRhiGraphicsPipeline::GreaterOrEqual:
6483 return MTLCompareFunctionGreaterEqual;
6484 case QRhiGraphicsPipeline::Always:
6485 return MTLCompareFunctionAlways;
6486 default:
6487 Q_UNREACHABLE();
6488 return MTLCompareFunctionAlways;
6489 }
6490}
6491
6492static inline MTLStencilOperation toMetalStencilOp(QRhiGraphicsPipeline::StencilOp op)
6493{
6494 switch (op) {
6495 case QRhiGraphicsPipeline::StencilZero:
6496 return MTLStencilOperationZero;
6497 case QRhiGraphicsPipeline::Keep:
6498 return MTLStencilOperationKeep;
6499 case QRhiGraphicsPipeline::Replace:
6500 return MTLStencilOperationReplace;
6501 case QRhiGraphicsPipeline::IncrementAndClamp:
6502 return MTLStencilOperationIncrementClamp;
6503 case QRhiGraphicsPipeline::DecrementAndClamp:
6504 return MTLStencilOperationDecrementClamp;
6505 case QRhiGraphicsPipeline::Invert:
6506 return MTLStencilOperationInvert;
6507 case QRhiGraphicsPipeline::IncrementAndWrap:
6508 return MTLStencilOperationIncrementWrap;
6509 case QRhiGraphicsPipeline::DecrementAndWrap:
6510 return MTLStencilOperationDecrementWrap;
6511 default:
6512 Q_UNREACHABLE();
6513 return MTLStencilOperationKeep;
6514 }
6515}
6516
6517static inline MTLPrimitiveType toMetalPrimitiveType(QRhiGraphicsPipeline::Topology t)
6518{
6519 switch (t) {
6520 case QRhiGraphicsPipeline::Triangles:
6521 return MTLPrimitiveTypeTriangle;
6522 case QRhiGraphicsPipeline::TriangleStrip:
6523 return MTLPrimitiveTypeTriangleStrip;
6524 case QRhiGraphicsPipeline::Lines:
6525 return MTLPrimitiveTypeLine;
6526 case QRhiGraphicsPipeline::LineStrip:
6527 return MTLPrimitiveTypeLineStrip;
6528 case QRhiGraphicsPipeline::Points:
6529 return MTLPrimitiveTypePoint;
6530 default:
6531 Q_UNREACHABLE();
6532 return MTLPrimitiveTypeTriangle;
6533 }
6534}
6535
6536static inline MTLPrimitiveTopologyClass toMetalPrimitiveTopologyClass(QRhiGraphicsPipeline::Topology t)
6537{
6538 switch (t) {
6539 case QRhiGraphicsPipeline::Triangles:
6540 case QRhiGraphicsPipeline::TriangleStrip:
6541 case QRhiGraphicsPipeline::TriangleFan:
6542 return MTLPrimitiveTopologyClassTriangle;
6543 case QRhiGraphicsPipeline::Lines:
6544 case QRhiGraphicsPipeline::LineStrip:
6545 return MTLPrimitiveTopologyClassLine;
6546 case QRhiGraphicsPipeline::Points:
6547 return MTLPrimitiveTopologyClassPoint;
6548 default:
6549 Q_UNREACHABLE();
6550 return MTLPrimitiveTopologyClassTriangle;
6551 }
6552}
6553
6554static inline MTLCullMode toMetalCullMode(QRhiGraphicsPipeline::CullMode c)
6555{
6556 switch (c) {
6557 case QRhiGraphicsPipeline::None:
6558 return MTLCullModeNone;
6559 case QRhiGraphicsPipeline::Front:
6560 return MTLCullModeFront;
6561 case QRhiGraphicsPipeline::Back:
6562 return MTLCullModeBack;
6563 default:
6564 Q_UNREACHABLE();
6565 return MTLCullModeNone;
6566 }
6567}
6568
6569static inline MTLTriangleFillMode toMetalTriangleFillMode(QRhiGraphicsPipeline::PolygonMode mode)
6570{
6571 switch (mode) {
6572 case QRhiGraphicsPipeline::Fill:
6573 return MTLTriangleFillModeFill;
6574 case QRhiGraphicsPipeline::Line:
6575 return MTLTriangleFillModeLines;
6576 default:
6577 Q_UNREACHABLE();
6578 return MTLTriangleFillModeFill;
6579 }
6580}
6581
6582static inline MTLWinding toMetalTessellationWindingOrder(QShaderDescription::TessellationWindingOrder w)
6583{
6584 switch (w) {
6585 case QShaderDescription::CwTessellationWindingOrder:
6586 return MTLWindingClockwise;
6587 case QShaderDescription::CcwTessellationWindingOrder:
6588 return MTLWindingCounterClockwise;
6589 default:
6590 // this is reachable, consider a tess.eval. shader not declaring it, the value is then Unknown
6591 return MTLWindingCounterClockwise;
6592 }
6593}
6594
6595static inline MTLTessellationPartitionMode toMetalTessellationPartitionMode(QShaderDescription::TessellationPartitioning p)
6596{
6597 switch (p) {
6598 case QShaderDescription::EqualTessellationPartitioning:
6599 return MTLTessellationPartitionModePow2;
6600 case QShaderDescription::FractionalEvenTessellationPartitioning:
6601 return MTLTessellationPartitionModeFractionalEven;
6602 case QShaderDescription::FractionalOddTessellationPartitioning:
6603 return MTLTessellationPartitionModeFractionalOdd;
6604 default:
6605 // this is reachable, consider a tess.eval. shader not declaring it, the value is then Unknown
6606 return MTLTessellationPartitionModePow2;
6607 }
6608}
6609
6610static inline MTLLanguageVersion toMetalLanguageVersion(const QShaderVersion &version)
6611{
6612 int v = version.version();
6613 return MTLLanguageVersion(((v / 10) << 16) + (v % 10));
6614}
6615
6616id<MTLBuffer> QRhiMetalData::allocFromStagingArea(StagingArea *area, quint32 size, quint32 alignment,
6617 int frameSlot, quint32 minBlockSize, quint32 *offset)
6618{
6619 auto &pool(*area);
6620 const quint32 alignedSize = aligned<quint32>(size, alignment);
6621 if (pool.offset + alignedSize > pool.capacity) {
6622 if (pool.buf) {
6623 // Reusing the StagingBuffer entry type: the handler just releases
6624 // the buffer. lastActiveFrameSlot = frameSlot is what keeps the old
6625 // buffer alive for the rest of this frame, since
6626 // executeDeferredReleases() only gets to this slot after the
6627 // semaphore wait in the next beginFrame() for it.
6630 e.lastActiveFrameSlot = frameSlot;
6631 e.stagingBuffer.buffer = pool.buf;
6632 releaseQueue.append(e);
6633 }
6634 pool.capacity = qMax(pool.capacity * 2, qMax(alignedSize, minBlockSize));
6635 pool.buf = [dev newBufferWithLength: pool.capacity options: MTLResourceStorageModeShared];
6636 pool.offset = 0;
6637 if (!pool.buf) {
6638 pool.capacity = 0;
6639 return nil;
6640 }
6641 }
6642 *offset = pool.offset;
6643 pool.offset += alignedSize;
6644 return pool.buf;
6645}
6646
6647id<MTLBuffer> QRhiMetalData::allocArgumentBuffer(quint32 size, quint32 alignment, int frameSlot, quint32 *offset)
6648{
6649 return allocFromStagingArea(&argBufPool[frameSlot], size, alignment, frameSlot, 16384, offset);
6650}
6651
6653{
6654 id<MTLBuffer> buf = [dev newBufferWithLength: size options: MTLResourceStorageModeShared];
6655 if (!buf)
6656 return nil;
6659 e.lastActiveFrameSlot = frameSlot;
6660 e.stagingBuffer.buffer = buf;
6661 releaseQueue.append(e);
6662 return buf;
6663}
6664
6665id<MTLBuffer> QRhiMetalData::allocBufferStaging(quint32 size, int frameSlot, quint32 *offset)
6666{
6667 if (size > LARGE_STAGING_ALLOC) {
6668 *offset = 0;
6669 return newOneShotStagingBuffer(size, frameSlot);
6670 }
6671
6672 StagingArea &area(bufStagingPool[frameSlot]);
6673 // 4 byte alignment is what MTLBlitCommandEncoder wants on non-Apple GPUs.
6674 const quint32 alignedSize = aligned<quint32>(size, 4u);
6675 area.bytesNeeded += alignedSize;
6676
6677 if (area.offset + alignedSize > area.capacity) {
6678 *offset = 0;
6679 return newOneShotStagingBuffer(size, frameSlot);
6680 }
6681
6682 *offset = area.offset;
6683 area.offset += alignedSize;
6684 return area.buf;
6685}
6686
6688{
6689 StagingArea &area(bufStagingPool[frameSlot]);
6690 const quint32 needed = area.bytesNeeded;
6691 area.bytesNeeded = 0;
6692 area.offset = 0;
6693
6694 bool resize = false;
6695 quint32 newCapacity = 0;
6696 if (needed > area.capacity) {
6697 // Fell back to one-shot buffers last time. Only grow once the demand
6698 // has proven to be recurring: a one-time burst, which is what loading a
6699 // scene looks like, would otherwise size the area permanently right at
6700 // the point where it is not needed any more.
6701 area.lowDemandFrames = 0;
6703 newCapacity = qMin(qNextPowerOfTwo(needed), STAGING_AREA_MAX);
6704 resize = true;
6705 }
6706 } else if (needed <= area.capacity / 4 && area.capacity > 0) {
6707 area.highDemandFrames = 0;
6709 area.lowDemandFrames = 0;
6710 // Nothing at all for a while, e.g. an app that only uploads during
6711 // startup: give the memory back entirely.
6712 newCapacity = needed ? qMax(area.capacity / 2, STAGING_AREA_MIN) : 0;
6713 resize = true;
6714 }
6715 } else {
6716 area.lowDemandFrames = 0;
6717 area.highDemandFrames = 0;
6718 }
6719
6720 if (resize && newCapacity != area.capacity) {
6721 if (area.buf) {
6724 e.lastActiveFrameSlot = frameSlot;
6725 e.stagingBuffer.buffer = area.buf;
6726 releaseQueue.append(e);
6727 area.buf = nil;
6728 }
6729 area.capacity = 0;
6730 if (newCapacity) {
6731 area.buf = [dev newBufferWithLength: newCapacity options: MTLResourceStorageModeShared];
6732 if (area.buf)
6733 area.capacity = newCapacity;
6734 }
6735 }
6736}
6737
6738id<MTLLibrary> QRhiMetalData::createMetalLib(const QShader &shader, QShader::Variant shaderVariant,
6739 bool preferArgumentBuffers,
6740 QString *error, QByteArray *entryPoint, QShaderKey *activeKey)
6741{
6742 QVarLengthArray<int, 8> versions;
6743 versions << 30 << 24 << 23 << 22 << 21 << 20 << 12;
6744
6745 // preferArgumentBuffers overrides whatever variant is set in the QShader.
6746 // This is by design, since we cannot expect the client to start specifying
6747 // the ArgumentBuffer variant that is only relevant for Metal. Also not
6748 // compatible with BatchableVertexShader and other variants since
6749 // ArgumentBufferShader is expected to be the argument-buffers version of
6750 // StandardShader, and it cannot be a combination of multiple variants. That
6751 // limitation should be fine for now.
6752
6753 QVarLengthArray<QShader::Variant, 2> variants;
6754 if (preferArgumentBuffers)
6755 variants << QShader::ArgumentBufferShader;
6756 variants << shaderVariant;
6757
6758 const QList<QShaderKey> shaders = shader.availableShaders();
6759
6760 auto findKey = [&shaders, &versions, &variants](QShader::Source source, QShaderKey *result) {
6761 for (const QShader::Variant &variant : variants) {
6762 for (const int &version : versions) {
6763 const QShaderKey key = { source, version, variant };
6764 if (shaders.contains(key)) {
6765 *result = key;
6766 return true;
6767 }
6768 }
6769 }
6770 return false;
6771 };
6772
6773 QShaderKey key;
6774
6775 if (findKey(QShader::Source::MetalLibShader, &key)) {
6776 QShaderCode mtllib = shader.shader(key);
6777 dispatch_data_t data = dispatch_data_create(mtllib.shader().constData(),
6778 size_t(mtllib.shader().size()),
6779 dispatch_get_global_queue(0, 0),
6780 DISPATCH_DATA_DESTRUCTOR_DEFAULT);
6781 NSError *err = nil;
6782 id<MTLLibrary> lib = [dev newLibraryWithData: data error: &err];
6783 dispatch_release(data);
6784 if (!err) {
6785 *entryPoint = mtllib.entryPoint();
6786 *activeKey = key;
6787 return lib;
6788 } else {
6789 const QString msg = QString::fromNSString(err.localizedDescription);
6790 qWarning("Failed to load metallib from baked shader: %s", qPrintable(msg));
6791 }
6792 }
6793
6794 if (!findKey(QShader::Source::MslShader, &key)) {
6795 qWarning() << "No MSL code found in baked shader" << shader;
6796 return nil;
6797 }
6798
6799 QShaderCode mslSource = shader.shader(key);
6800
6801 NSString *src = [NSString stringWithUTF8String: mslSource.shader().constData()];
6802 MTLCompileOptions *opts = [[MTLCompileOptions alloc] init];
6803 opts.languageVersion = toMetalLanguageVersion(key.sourceVersion());
6804 NSError *err = nil;
6805 id<MTLLibrary> lib = [dev newLibraryWithSource: src options: opts error: &err];
6806 [opts release];
6807 // src is autoreleased
6808
6809 // if lib is null and err is non-null, we had errors (fail)
6810 // if lib is non-null and err is non-null, we had warnings (success)
6811 // if lib is non-null and err is null, there were no errors or warnings (success)
6812 if (!lib) {
6813 const QString msg = QString::fromNSString(err.localizedDescription);
6814 *error = msg;
6815 return nil;
6816 }
6817
6818 *entryPoint = mslSource.entryPoint();
6819 *activeKey = key;
6820 return lib;
6821}
6822
6823id<MTLFunction> QRhiMetalData::createMSLShaderFunction(id<MTLLibrary> lib, const QByteArray &entryPoint)
6824{
6825 return [lib newFunctionWithName:[NSString stringWithUTF8String:entryPoint.constData()]];
6826}
6827
6829{
6830 MTLRenderPipelineDescriptor *rpDesc = reinterpret_cast<MTLRenderPipelineDescriptor *>(metalRpDesc);
6831
6832 if (rpD->colorAttachmentCount) {
6833 // defaults when no targetBlends are provided
6834 rpDesc.colorAttachments[0].pixelFormat = MTLPixelFormat(rpD->colorFormat[0]);
6835 rpDesc.colorAttachments[0].writeMask = MTLColorWriteMaskAll;
6836 rpDesc.colorAttachments[0].blendingEnabled = false;
6837
6838 Q_ASSERT(m_targetBlends.count() == rpD->colorAttachmentCount
6839 || (m_targetBlends.isEmpty() && rpD->colorAttachmentCount == 1));
6840
6841 for (uint i = 0, ie = uint(m_targetBlends.count()); i != ie; ++i) {
6842 const QRhiGraphicsPipeline::TargetBlend &b(m_targetBlends[int(i)]);
6843 rpDesc.colorAttachments[i].pixelFormat = MTLPixelFormat(rpD->colorFormat[i]);
6844 rpDesc.colorAttachments[i].blendingEnabled = b.enable;
6845 rpDesc.colorAttachments[i].sourceRGBBlendFactor = toMetalBlendFactor(b.srcColor);
6846 rpDesc.colorAttachments[i].destinationRGBBlendFactor = toMetalBlendFactor(b.dstColor);
6847 rpDesc.colorAttachments[i].rgbBlendOperation = toMetalBlendOp(b.opColor);
6848 rpDesc.colorAttachments[i].sourceAlphaBlendFactor = toMetalBlendFactor(b.srcAlpha);
6849 rpDesc.colorAttachments[i].destinationAlphaBlendFactor = toMetalBlendFactor(b.dstAlpha);
6850 rpDesc.colorAttachments[i].alphaBlendOperation = toMetalBlendOp(b.opAlpha);
6851 rpDesc.colorAttachments[i].writeMask = toMetalColorWriteMask(b.colorWrite);
6852 }
6853 }
6854
6855 if (rpD->hasDepthStencil) {
6856 // Must only be set when a depth-stencil buffer will actually be bound,
6857 // validation blows up otherwise.
6858 MTLPixelFormat fmt = MTLPixelFormat(rpD->dsFormat);
6859 rpDesc.depthAttachmentPixelFormat = fmt;
6860#if defined(Q_OS_MACOS)
6861 if (fmt != MTLPixelFormatDepth16Unorm && fmt != MTLPixelFormatDepth32Float)
6862#else
6863 if (fmt != MTLPixelFormatDepth32Float)
6864#endif
6865 rpDesc.stencilAttachmentPixelFormat = fmt;
6866 }
6867
6868 QRHI_RES_RHI(QRhiMetal);
6869 rpDesc.rasterSampleCount = NSUInteger(rhiD->effectiveSampleCount(m_sampleCount));
6870}
6871
6873{
6874 MTLDepthStencilDescriptor *dsDesc = reinterpret_cast<MTLDepthStencilDescriptor *>(metalDsDesc);
6875
6876 dsDesc.depthCompareFunction = m_depthTest ? toMetalCompareOp(m_depthOp) : MTLCompareFunctionAlways;
6877 dsDesc.depthWriteEnabled = m_depthWrite;
6878 if (m_stencilTest) {
6879 dsDesc.frontFaceStencil = [[MTLStencilDescriptor alloc] init];
6880 dsDesc.frontFaceStencil.stencilFailureOperation = toMetalStencilOp(m_stencilFront.failOp);
6881 dsDesc.frontFaceStencil.depthFailureOperation = toMetalStencilOp(m_stencilFront.depthFailOp);
6882 dsDesc.frontFaceStencil.depthStencilPassOperation = toMetalStencilOp(m_stencilFront.passOp);
6883 dsDesc.frontFaceStencil.stencilCompareFunction = toMetalCompareOp(m_stencilFront.compareOp);
6884 dsDesc.frontFaceStencil.readMask = m_stencilReadMask;
6885 dsDesc.frontFaceStencil.writeMask = m_stencilWriteMask;
6886
6887 dsDesc.backFaceStencil = [[MTLStencilDescriptor alloc] init];
6888 dsDesc.backFaceStencil.stencilFailureOperation = toMetalStencilOp(m_stencilBack.failOp);
6889 dsDesc.backFaceStencil.depthFailureOperation = toMetalStencilOp(m_stencilBack.depthFailOp);
6890 dsDesc.backFaceStencil.depthStencilPassOperation = toMetalStencilOp(m_stencilBack.passOp);
6891 dsDesc.backFaceStencil.stencilCompareFunction = toMetalCompareOp(m_stencilBack.compareOp);
6892 dsDesc.backFaceStencil.readMask = m_stencilReadMask;
6893 dsDesc.backFaceStencil.writeMask = m_stencilWriteMask;
6894 }
6895}
6896
6898{
6899 d->winding = m_frontFace == CCW ? MTLWindingCounterClockwise : MTLWindingClockwise;
6900 d->cullMode = toMetalCullMode(m_cullMode);
6901 d->triangleFillMode = toMetalTriangleFillMode(m_polygonMode);
6902 d->depthClipMode = m_depthClamp ? MTLDepthClipModeClamp : MTLDepthClipModeClip;
6903 d->depthBias = float(m_depthBias);
6904 d->slopeScaledDepthBias = m_slopeScaledDepthBias;
6905}
6906
6908{
6909 // same binding space for vertex and constant buffers - work it around
6910 // should be in native resource binding not SPIR-V, but this will work anyway
6911 const int firstVertexBinding = QRHI_RES(QMetalShaderResourceBindings, q->shaderResourceBindings())->maxBinding + 1;
6912
6913 QRhiVertexInputLayout vertexInputLayout = q->vertexInputLayout();
6914 for (auto it = vertexInputLayout.cbeginAttributes(), itEnd = vertexInputLayout.cendAttributes();
6915 it != itEnd; ++it)
6916 {
6917 const uint loc = uint(it->location());
6918 desc.attributes[loc].format = decltype(desc.attributes[loc].format)(toMetalAttributeFormat(it->format()));
6919 desc.attributes[loc].offset = NSUInteger(it->offset());
6920 desc.attributes[loc].bufferIndex = NSUInteger(firstVertexBinding + it->binding());
6921 }
6922 int bindingIndex = 0;
6923 const NSUInteger viewCount = qMax<NSUInteger>(1, q->multiViewCount());
6924 for (auto it = vertexInputLayout.cbeginBindings(), itEnd = vertexInputLayout.cendBindings();
6925 it != itEnd; ++it, ++bindingIndex)
6926 {
6927 const uint layoutIdx = uint(firstVertexBinding + bindingIndex);
6928 desc.layouts[layoutIdx].stepFunction =
6929 it->classification() == QRhiVertexInputBinding::PerInstance
6930 ? MTLVertexStepFunctionPerInstance : MTLVertexStepFunctionPerVertex;
6931 desc.layouts[layoutIdx].stepRate = NSUInteger(it->instanceStepRate());
6932 if (desc.layouts[layoutIdx].stepFunction == MTLVertexStepFunctionPerInstance)
6933 desc.layouts[layoutIdx].stepRate *= viewCount;
6934 desc.layouts[layoutIdx].stride = it->stride();
6935 }
6936}
6937
6938void QMetalGraphicsPipelineData::setupStageInputDescriptor(MTLStageInputOutputDescriptor *desc)
6939{
6940 // same binding space for vertex and constant buffers - work it around
6941 // should be in native resource binding not SPIR-V, but this will work anyway
6942 const int firstVertexBinding = QRHI_RES(QMetalShaderResourceBindings, q->shaderResourceBindings())->maxBinding + 1;
6943
6944 QRhiVertexInputLayout vertexInputLayout = q->vertexInputLayout();
6945 for (auto it = vertexInputLayout.cbeginAttributes(), itEnd = vertexInputLayout.cendAttributes();
6946 it != itEnd; ++it)
6947 {
6948 const uint loc = uint(it->location());
6949 desc.attributes[loc].format = decltype(desc.attributes[loc].format)(toMetalAttributeFormat(it->format()));
6950 desc.attributes[loc].offset = NSUInteger(it->offset());
6951 desc.attributes[loc].bufferIndex = NSUInteger(firstVertexBinding + it->binding());
6952 }
6953 int bindingIndex = 0;
6954 for (auto it = vertexInputLayout.cbeginBindings(), itEnd = vertexInputLayout.cendBindings();
6955 it != itEnd; ++it, ++bindingIndex)
6956 {
6957 const uint layoutIdx = uint(firstVertexBinding + bindingIndex);
6958 if (desc.indexBufferIndex) {
6959 desc.layouts[layoutIdx].stepFunction =
6960 it->classification() == QRhiVertexInputBinding::PerInstance
6961 ? MTLStepFunctionThreadPositionInGridY : MTLStepFunctionThreadPositionInGridXIndexed;
6962 } else {
6963 desc.layouts[layoutIdx].stepFunction =
6964 it->classification() == QRhiVertexInputBinding::PerInstance
6965 ? MTLStepFunctionThreadPositionInGridY : MTLStepFunctionThreadPositionInGridX;
6966 }
6967 desc.layouts[layoutIdx].stepRate = NSUInteger(it->instanceStepRate());
6968 desc.layouts[layoutIdx].stride = it->stride();
6969 }
6970}
6971
6972void QRhiMetalData::trySeedingRenderPipelineFromBinaryArchive(MTLRenderPipelineDescriptor *rpDesc)
6973{
6974 if (binArch) {
6975 NSArray *binArchArray = [NSArray arrayWithObjects: binArch, nil];
6976 rpDesc.binaryArchives = binArchArray;
6977 }
6978}
6979
6980void QRhiMetalData::addRenderPipelineToBinaryArchive(MTLRenderPipelineDescriptor *rpDesc)
6981{
6982 if (binArch) {
6983 NSError *err = nil;
6984 if (![binArch addRenderPipelineFunctionsWithDescriptor: rpDesc error: &err]) {
6985 const QString msg = QString::fromNSString(err.localizedDescription);
6986 qWarning("Failed to collect render pipeline functions to binary archive: %s", qPrintable(msg));
6987 }
6988 }
6989}
6990
6991static inline bool usesTextures(const QShaderDescription &desc)
6992{
6993 return !desc.combinedImageSamplers().isEmpty()
6994 || !desc.separateImages().isEmpty()
6995 || !desc.storageImages().isEmpty();
6996}
6997
6998static bool hasArgumentBufferVariant(const QShader &shader)
6999{
7000 for (const QShaderKey &k : shader.availableShaders()) {
7001 if (k.sourceVariant() == QShader::ArgumentBufferShader)
7002 return true;
7003 }
7004 return false;
7005}
7006
7008{
7009 const int index = shader->nativeShaderInfo.extraBufferBindings.value(QShaderPrivate::MslArgumentBufferBinding, -1);
7010 if (index >= 0) {
7011 shader->argumentBufferIndex = index;
7012 shader->argumentEncoder = [shader->func newArgumentEncoderWithBufferIndex: NSUInteger(index)];
7013 }
7014}
7015
7017{
7018 QRHI_RES_RHI(QRhiMetal);
7019
7020 // A pipeline that supports indirect command buffers cannot have textures
7021 // bound directly to its functions, so prefer the shader variant that
7022 // reaches them through an argument buffer.
7023 const bool wantArgumentBuffers = m_flags.testFlag(UsesIndirectDraws) && rhiD->caps.indirectCommandBuffers;
7024
7025 MTLVertexDescriptor *vertexDesc = [MTLVertexDescriptor vertexDescriptor];
7026 d->setupVertexInputDescriptor(vertexDesc);
7027
7028 MTLRenderPipelineDescriptor *rpDesc = [[MTLRenderPipelineDescriptor alloc] init];
7029 rpDesc.vertexDescriptor = vertexDesc;
7030
7031 // Mutability cannot be determined (slotted buffers could be set as
7032 // MTLMutabilityImmutable, but then we potentially need a different
7033 // descriptor for each buffer combination as this depends on the actual
7034 // buffers not just the resource binding layout), so leave
7035 // rpDesc.vertex/fragmentBuffers at the defaults.
7036
7037 for (const QRhiShaderStage &shaderStage : std::as_const(m_shaderStages)) {
7038 const QShader shader = shaderStage.shader();
7039 const bool argumentBufferBuild = wantArgumentBuffers && hasArgumentBufferVariant(shader);
7040 auto cacheIt = rhiD->d->shaderCache.constFind({ shaderStage, argumentBufferBuild });
7041 if (cacheIt != rhiD->d->shaderCache.constEnd()) {
7042 switch (shaderStage.type()) {
7043 case QRhiShaderStage::Vertex:
7044 d->vs = *cacheIt;
7045 [d->vs.lib retain];
7046 [d->vs.func retain];
7047 [d->vs.argumentEncoder retain];
7048 rpDesc.vertexFunction = d->vs.func;
7049 break;
7050 case QRhiShaderStage::Fragment:
7051 d->fs = *cacheIt;
7052 [d->fs.lib retain];
7053 [d->fs.func retain];
7054 [d->fs.argumentEncoder retain];
7055 rpDesc.fragmentFunction = d->fs.func;
7056 break;
7057 default:
7058 break;
7059 }
7060 } else {
7061 QString error;
7062 QByteArray entryPoint;
7063 QShaderKey activeKey;
7064 id<MTLLibrary> lib = rhiD->d->createMetalLib(shader, shaderStage.shaderVariant(),
7065 argumentBufferBuild,
7066 &error, &entryPoint, &activeKey);
7067 if (!lib) {
7068 qWarning("MSL shader compilation failed: %s", qPrintable(error));
7069 return false;
7070 }
7071 id<MTLFunction> func = rhiD->d->createMSLShaderFunction(lib, entryPoint);
7072 if (!func) {
7073 qWarning("MSL function for entry point %s not found", entryPoint.constData());
7074 [lib release];
7075 return false;
7076 }
7077 if (rhiD->d->shaderCache.count() >= QRhiMetal::MAX_SHADER_CACHE_ENTRIES) {
7078 // Use the simplest strategy: too many cached shaders -> drop them all.
7079 for (QMetalShader &s : rhiD->d->shaderCache)
7080 s.destroy();
7081 rhiD->d->shaderCache.clear();
7082 }
7083 switch (shaderStage.type()) {
7084 case QRhiShaderStage::Vertex:
7085 d->vs.lib = lib;
7086 d->vs.func = func;
7087 d->vs.nativeResourceBindingMap = shader.nativeResourceBindingMap(activeKey);
7088 d->vs.desc = shader.description();
7089 d->vs.nativeShaderInfo = shader.nativeShaderInfo(activeKey);
7090 setupArgumentBufferEncoder(&d->vs);
7091 rhiD->d->shaderCache.insert({ shaderStage, argumentBufferBuild }, d->vs);
7092 [d->vs.lib retain];
7093 [d->vs.func retain];
7094 [d->vs.argumentEncoder retain];
7095 rpDesc.vertexFunction = func;
7096 break;
7097 case QRhiShaderStage::Fragment:
7098 d->fs.lib = lib;
7099 d->fs.func = func;
7100 d->fs.nativeResourceBindingMap = shader.nativeResourceBindingMap(activeKey);
7101 d->fs.desc = shader.description();
7102 d->fs.nativeShaderInfo = shader.nativeShaderInfo(activeKey);
7103 setupArgumentBufferEncoder(&d->fs);
7104 rhiD->d->shaderCache.insert({ shaderStage, argumentBufferBuild }, d->fs);
7105 [d->fs.lib retain];
7106 [d->fs.func retain];
7107 [d->fs.argumentEncoder retain];
7108 rpDesc.fragmentFunction = func;
7109 break;
7110 default:
7111 [func release];
7112 [lib release];
7113 break;
7114 }
7115 }
7116 }
7117
7118 QMetalRenderPassDescriptor *rpD = QRHI_RES(QMetalRenderPassDescriptor, m_renderPassDesc);
7120
7121 // Safe for an indirect command buffer if the stage has no textures at all,
7122 // or reaches them through an argument buffer.
7123 const auto icbSafe = [](const QMetalShader &s) {
7124 return !usesTextures(s.desc) || s.argumentBufferIndex >= 0;
7125 };
7126 d->icbCapable = wantArgumentBuffers && icbSafe(d->vs) && icbSafe(d->fs);
7127 if (d->icbCapable)
7128 rpDesc.supportIndirectCommandBuffers = YES;
7129
7130 if (m_multiViewCount >= 2)
7131 rpDesc.inputPrimitiveTopology = toMetalPrimitiveTopologyClass(m_topology);
7132
7133 rhiD->d->trySeedingRenderPipelineFromBinaryArchive(rpDesc);
7134
7135 if (rhiD->rhiFlags.testFlag(QRhi::EnablePipelineCacheDataSave))
7136 rhiD->d->addRenderPipelineToBinaryArchive(rpDesc);
7137
7138 NSError *err = nil;
7139 d->ps = [rhiD->d->dev newRenderPipelineStateWithDescriptor: rpDesc error: &err];
7140 [rpDesc release];
7141 if (!d->ps) {
7142 const QString msg = QString::fromNSString(err.localizedDescription);
7143 qWarning("Failed to create render pipeline state: %s", qPrintable(msg));
7144 return false;
7145 }
7146
7147 MTLDepthStencilDescriptor *dsDesc = [[MTLDepthStencilDescriptor alloc] init];
7149 d->ds = [rhiD->d->dev newDepthStencilStateWithDescriptor: dsDesc];
7150 [dsDesc release];
7151
7152 d->primitiveType = toMetalPrimitiveType(m_topology);
7154
7155 return true;
7156}
7157
7158int QMetalGraphicsPipelineData::Tessellation::vsCompVariantToIndex(QShader::Variant vertexCompVariant)
7159{
7160 switch (vertexCompVariant) {
7161 case QShader::NonIndexedVertexAsComputeShader:
7162 return 0;
7163 case QShader::UInt32IndexedVertexAsComputeShader:
7164 return 1;
7165 case QShader::UInt16IndexedVertexAsComputeShader:
7166 return 2;
7167 default:
7168 break;
7169 }
7170 return -1;
7171}
7172
7174{
7175 const int varIndex = vsCompVariantToIndex(vertexCompVariant);
7176 if (varIndex >= 0 && vertexComputeState[varIndex])
7177 return vertexComputeState[varIndex];
7178
7179 id<MTLFunction> func = nil;
7180 if (varIndex >= 0)
7181 func = compVs[varIndex].func;
7182
7183 if (!func) {
7184 qWarning("No compute function found for vertex shader translated for tessellation, this should not happen");
7185 return nil;
7186 }
7187
7188 const QMap<int, int> &ebb(compVs[varIndex].nativeShaderInfo.extraBufferBindings);
7189 const int indexBufferBinding = ebb.value(QShaderPrivate::MslTessVertIndicesBufferBinding, -1);
7190
7191 MTLComputePipelineDescriptor *cpDesc = [MTLComputePipelineDescriptor new];
7192 cpDesc.computeFunction = func;
7193 cpDesc.threadGroupSizeIsMultipleOfThreadExecutionWidth = YES;
7194 cpDesc.stageInputDescriptor = [MTLStageInputOutputDescriptor stageInputOutputDescriptor];
7195 if (indexBufferBinding >= 0) {
7196 if (vertexCompVariant == QShader::UInt32IndexedVertexAsComputeShader) {
7197 cpDesc.stageInputDescriptor.indexType = MTLIndexTypeUInt32;
7198 cpDesc.stageInputDescriptor.indexBufferIndex = indexBufferBinding;
7199 } else if (vertexCompVariant == QShader::UInt16IndexedVertexAsComputeShader) {
7200 cpDesc.stageInputDescriptor.indexType = MTLIndexTypeUInt16;
7201 cpDesc.stageInputDescriptor.indexBufferIndex = indexBufferBinding;
7202 }
7203 }
7204 q->setupStageInputDescriptor(cpDesc.stageInputDescriptor);
7205
7206 rhiD->d->trySeedingComputePipelineFromBinaryArchive(cpDesc);
7207
7208 if (rhiD->rhiFlags.testFlag(QRhi::EnablePipelineCacheDataSave))
7209 rhiD->d->addComputePipelineToBinaryArchive(cpDesc);
7210
7211 NSError *err = nil;
7212 id<MTLComputePipelineState> ps = [rhiD->d->dev newComputePipelineStateWithDescriptor: cpDesc
7213 options: MTLPipelineOptionNone
7214 reflection: nil
7215 error: &err];
7216 [cpDesc release];
7217 if (!ps) {
7218 const QString msg = QString::fromNSString(err.localizedDescription);
7219 qWarning("Failed to create compute pipeline state: %s", qPrintable(msg));
7220 } else {
7221 vertexComputeState[varIndex] = ps;
7222 }
7223 // not retained, the only owner is vertexComputeState and so the QRhiGraphicsPipeline
7224 return ps;
7225}
7226
7228{
7229 if (tessControlComputeState)
7230 return tessControlComputeState;
7231
7232 MTLComputePipelineDescriptor *cpDesc = [MTLComputePipelineDescriptor new];
7233 cpDesc.computeFunction = compTesc.func;
7234
7235 rhiD->d->trySeedingComputePipelineFromBinaryArchive(cpDesc);
7236
7237 if (rhiD->rhiFlags.testFlag(QRhi::EnablePipelineCacheDataSave))
7238 rhiD->d->addComputePipelineToBinaryArchive(cpDesc);
7239
7240 NSError *err = nil;
7241 id<MTLComputePipelineState> ps = [rhiD->d->dev newComputePipelineStateWithDescriptor: cpDesc
7242 options: MTLPipelineOptionNone
7243 reflection: nil
7244 error: &err];
7245 [cpDesc release];
7246 if (!ps) {
7247 const QString msg = QString::fromNSString(err.localizedDescription);
7248 qWarning("Failed to create compute pipeline state: %s", qPrintable(msg));
7249 } else {
7250 tessControlComputeState = ps;
7251 }
7252 // not retained, the only owner is tessControlComputeState and so the QRhiGraphicsPipeline
7253 return ps;
7254}
7255
7256static inline bool indexTaken(quint32 index, quint64 indices)
7257{
7258 return (indices >> index) & 0x1;
7259}
7260
7261static inline void takeIndex(quint32 index, quint64 &indices)
7262{
7263 indices |= 1 << index;
7264}
7265
7266static inline int nextAttributeIndex(quint64 indices)
7267{
7268 // Maximum number of vertex attributes per vertex descriptor. There does
7269 // not appear to be a way to query this from the implementation.
7270 // https://developer.apple.com/metal/Metal-Feature-Set-Tables.pdf indicates
7271 // that all GPU families have a value of 31.
7272 static const int maxVertexAttributes = 31;
7273
7274 for (int index = 0; index < maxVertexAttributes; ++index) {
7275 if (!indexTaken(index, indices))
7276 return index;
7277 }
7278
7279 Q_UNREACHABLE_RETURN(-1);
7280}
7281
7282static inline int aligned(quint32 offset, quint32 alignment)
7283{
7284 return ((offset + alignment - 1) / alignment) * alignment;
7285}
7286
7287template<typename T>
7288static void addUnusedVertexAttribute(const T &variable, QRhiMetal *rhiD, quint32 &offset, quint32 &vertexAlignment)
7289{
7290
7291 int elements = 1;
7292 for (const int dim : variable.arrayDims)
7293 elements *= dim;
7294
7295 if (variable.type == QShaderDescription::VariableType::Struct) {
7296 for (int element = 0; element < elements; ++element) {
7297 for (const auto &member : variable.structMembers) {
7298 addUnusedVertexAttribute(member, rhiD, offset, vertexAlignment);
7299 }
7300 }
7301 } else {
7302 const QRhiVertexInputAttribute::Format format = rhiD->shaderDescVariableFormatToVertexInputFormat(variable.type);
7303 const quint32 size = rhiD->byteSizePerVertexForVertexInputFormat(format);
7304
7305 // MSL specification 3.0 says alignment = size for non packed scalars and vectors
7306 const quint32 alignment = size;
7307 vertexAlignment = std::max(vertexAlignment, alignment);
7308
7309 for (int element = 0; element < elements; ++element) {
7310 // adjust alignment
7311 offset = aligned(offset, alignment);
7312 offset += size;
7313 }
7314 }
7315}
7316
7317template<typename T>
7318static void addVertexAttribute(const T &variable, int binding, QRhiMetal *rhiD, int &index, quint32 &offset, MTLVertexAttributeDescriptorArray *attributes, quint64 &indices, quint32 &vertexAlignment)
7319{
7320
7321 int elements = 1;
7322 for (const int dim : variable.arrayDims)
7323 elements *= dim;
7324
7325 if (variable.type == QShaderDescription::VariableType::Struct) {
7326 for (int element = 0; element < elements; ++element) {
7327 for (const auto &member : variable.structMembers) {
7328 addVertexAttribute(member, binding, rhiD, index, offset, attributes, indices, vertexAlignment);
7329 }
7330 }
7331 } else {
7332 const QRhiVertexInputAttribute::Format format = rhiD->shaderDescVariableFormatToVertexInputFormat(variable.type);
7333 const quint32 size = rhiD->byteSizePerVertexForVertexInputFormat(format);
7334
7335 // MSL specification 3.0 says alignment = size for non packed scalars and vectors
7336 const quint32 alignment = size;
7337 vertexAlignment = std::max(vertexAlignment, alignment);
7338
7339 for (int element = 0; element < elements; ++element) {
7340 Q_ASSERT(!indexTaken(index, indices));
7341
7342 // adjust alignment
7343 offset = aligned(offset, alignment);
7344
7345 attributes[index].bufferIndex = binding;
7346 attributes[index].format = toMetalAttributeFormat(format);
7347 attributes[index].offset = offset;
7348
7349 takeIndex(index, indices);
7350 index++;
7351 if (indexTaken(index, indices))
7352 index = nextAttributeIndex(indices);
7353
7354 offset += size;
7355 }
7356 }
7357}
7358
7359static inline bool matches(const QList<QShaderDescription::BlockVariable> &a, const QList<QShaderDescription::BlockVariable> &b)
7360{
7361 if (a.size() == b.size()) {
7362 bool match = true;
7363 for (int i = 0; i < a.size() && match; ++i) {
7364 match &= a[i].type == b[i].type
7365 && a[i].arrayDims == b[i].arrayDims
7366 && matches(a[i].structMembers, b[i].structMembers);
7367 }
7368 return match;
7369 }
7370
7371 return false;
7372}
7373
7374static inline bool matches(const QShaderDescription::InOutVariable &a, const QShaderDescription::InOutVariable &b)
7375{
7376 return a.location == b.location
7377 && a.type == b.type
7378 && a.perPatch == b.perPatch
7379 && matches(a.structMembers, b.structMembers);
7380}
7381
7382//
7383// Create the tessellation evaluation render pipeline state
7384//
7385// The tesc runs as a compute shader in a compute pipeline and writes per patch and per patch
7386// control point data into separate storage buffers. The tese runs as a vertex shader in a render
7387// pipeline. Our task is to generate a render pipeline descriptor for the tese that pulls vertices
7388// from these buffers.
7389//
7390// As the buffers we are pulling vertices from are written by a compute pipeline, they follow the
7391// MSL alignment conventions which we must take into account when generating our
7392// MTLVertexDescriptor. We must include the user defined tese input attributes, and any builtins
7393// that were used.
7394//
7395// SPIRV-Cross generates the MSL tese shader code with input attribute indices that reflect the
7396// specified GLSL locations. Interface blocks are flattened with each member having an incremented
7397// attribute index. SPIRV-Cross reports an error on compilation if there are clashes in the index
7398// address space.
7399//
7400// After the user specified attributes are processed, SPIRV-Cross places the in-use builtins at the
7401// next available (lowest value) attribute index. Tese builtins are processed in the following
7402// order:
7403//
7404// in gl_PerVertex
7405// {
7406// vec4 gl_Position;
7407// float gl_PointSize;
7408// float gl_ClipDistance[];
7409// };
7410//
7411// patch in float gl_TessLevelOuter[4];
7412// patch in float gl_TessLevelInner[2];
7413//
7414// Enumerations in QShaderDescription::BuiltinType are defined in this order.
7415//
7416// For quads, SPIRV-Cross places MTLQuadTessellationFactorsHalf per patch in the tessellation
7417// factor buffer. For triangles it uses MTLTriangleTessellationFactorsHalf.
7418//
7419// It should be noted that SPIRV-Cross handles the following builtin inputs internally, with no
7420// host side support required.
7421//
7422// in vec3 gl_TessCoord;
7423// in int gl_PatchVerticesIn;
7424// in int gl_PrimitiveID;
7425//
7427{
7428 if (pipeline->d->ps)
7429 return pipeline->d->ps;
7430
7431 MTLRenderPipelineDescriptor *rpDesc = [[MTLRenderPipelineDescriptor alloc] init];
7432 MTLVertexDescriptor *vertexDesc = [MTLVertexDescriptor vertexDescriptor];
7433
7434 // tesc output buffers
7435 const QMap<int, int> &ebb(compTesc.nativeShaderInfo.extraBufferBindings);
7436 const int tescOutputBufferBinding = ebb.value(QShaderPrivate::MslTessVertTescOutputBufferBinding, -1);
7437 const int tescPatchOutputBufferBinding = ebb.value(QShaderPrivate::MslTessTescPatchOutputBufferBinding, -1);
7438 const int tessFactorBufferBinding = ebb.value(QShaderPrivate::MslTessTescTessLevelBufferBinding, -1);
7439 quint32 offsetInTescOutput = 0;
7440 quint32 offsetInTescPatchOutput = 0;
7441 quint32 offsetInTessFactorBuffer = 0;
7442 quint32 tescOutputAlignment = 0;
7443 quint32 tescPatchOutputAlignment = 0;
7444 quint32 tessFactorAlignment = 0;
7445 QSet<int> usedBuffers;
7446
7447 // tesc output variables in ascending location order
7448 QMap<int, QShaderDescription::InOutVariable> tescOutVars;
7449 for (const auto &tescOutVar : compTesc.desc.outputVariables())
7450 tescOutVars[tescOutVar.location] = tescOutVar;
7451
7452 // tese input variables in ascending location order
7453 QMap<int, QShaderDescription::InOutVariable> teseInVars;
7454 for (const auto &teseInVar : vertTese.desc.inputVariables())
7455 teseInVars[teseInVar.location] = teseInVar;
7456
7457 // bit mask tracking usage of vertex attribute indices
7458 quint64 indices = 0;
7459
7460 for (QShaderDescription::InOutVariable &tescOutVar : tescOutVars) {
7461
7462 int index = tescOutVar.location;
7463 int binding = -1;
7464 quint32 *offset = nullptr;
7465 quint32 *alignment = nullptr;
7466
7467 if (tescOutVar.perPatch) {
7468 binding = tescPatchOutputBufferBinding;
7469 offset = &offsetInTescPatchOutput;
7470 alignment = &tescPatchOutputAlignment;
7471 } else {
7472 tescOutVar.arrayDims.removeLast();
7473 binding = tescOutputBufferBinding;
7474 offset = &offsetInTescOutput;
7475 alignment = &tescOutputAlignment;
7476 }
7477
7478 if (teseInVars.contains(index)) {
7479
7480 if (!matches(teseInVars[index], tescOutVar)) {
7481 qWarning() << "mismatched tessellation control output -> tesssellation evaluation input at location" << index;
7482 qWarning() << " tesc out:" << tescOutVar;
7483 qWarning() << " tese in:" << teseInVars[index];
7484 }
7485
7486 if (binding != -1) {
7487 addVertexAttribute(tescOutVar, binding, rhiD, index, *offset, vertexDesc.attributes, indices, *alignment);
7488 usedBuffers << binding;
7489 } else {
7490 qWarning() << "baked tessellation control shader missing output buffer binding information";
7491 addUnusedVertexAttribute(tescOutVar, rhiD, *offset, *alignment);
7492 }
7493
7494 } else {
7495 qWarning() << "missing tessellation evaluation input for tessellation control output:" << tescOutVar;
7496 addUnusedVertexAttribute(tescOutVar, rhiD, *offset, *alignment);
7497 }
7498
7499 teseInVars.remove(tescOutVar.location);
7500 }
7501
7502 for (const QShaderDescription::InOutVariable &teseInVar : teseInVars)
7503 qWarning() << "missing tessellation control output for tessellation evaluation input:" << teseInVar;
7504
7505 // tesc output builtins in ascending location order
7506 QMap<QShaderDescription::BuiltinType, QShaderDescription::BuiltinVariable> tescOutBuiltins;
7507 for (const auto &tescOutBuiltin : compTesc.desc.outputBuiltinVariables())
7508 tescOutBuiltins[tescOutBuiltin.type] = tescOutBuiltin;
7509
7510 // tese input builtins in ascending location order
7511 QMap<QShaderDescription::BuiltinType, QShaderDescription::BuiltinVariable> teseInBuiltins;
7512 for (const auto &teseInBuiltin : vertTese.desc.inputBuiltinVariables())
7513 teseInBuiltins[teseInBuiltin.type] = teseInBuiltin;
7514
7515 const bool trianglesMode = vertTese.desc.tessellationMode() == QShaderDescription::TrianglesTessellationMode;
7516 bool tessLevelAdded = false;
7517
7518 for (const QShaderDescription::BuiltinVariable &builtin : tescOutBuiltins) {
7519
7520 QShaderDescription::InOutVariable variable;
7521 int binding = -1;
7522 quint32 *offset = nullptr;
7523 quint32 *alignment = nullptr;
7524
7525 switch (builtin.type) {
7526 case QShaderDescription::BuiltinType::PositionBuiltin:
7527 variable.type = QShaderDescription::VariableType::Vec4;
7528 binding = tescOutputBufferBinding;
7529 offset = &offsetInTescOutput;
7530 alignment = &tescOutputAlignment;
7531 break;
7532 case QShaderDescription::BuiltinType::PointSizeBuiltin:
7533 variable.type = QShaderDescription::VariableType::Float;
7534 binding = tescOutputBufferBinding;
7535 offset = &offsetInTescOutput;
7536 alignment = &tescOutputAlignment;
7537 break;
7538 case QShaderDescription::BuiltinType::ClipDistanceBuiltin:
7539 variable.type = QShaderDescription::VariableType::Float;
7540 variable.arrayDims = builtin.arrayDims;
7541 binding = tescOutputBufferBinding;
7542 offset = &offsetInTescOutput;
7543 alignment = &tescOutputAlignment;
7544 break;
7545 case QShaderDescription::BuiltinType::TessLevelOuterBuiltin:
7546 variable.type = QShaderDescription::VariableType::Half4;
7547 binding = tessFactorBufferBinding;
7548 offset = &offsetInTessFactorBuffer;
7549 tessLevelAdded = trianglesMode;
7550 alignment = &tessFactorAlignment;
7551 break;
7552 case QShaderDescription::BuiltinType::TessLevelInnerBuiltin:
7553 if (trianglesMode) {
7554 if (!tessLevelAdded) {
7555 variable.type = QShaderDescription::VariableType::Half4;
7556 binding = tessFactorBufferBinding;
7557 offsetInTessFactorBuffer = 0;
7558 offset = &offsetInTessFactorBuffer;
7559 alignment = &tessFactorAlignment;
7560 tessLevelAdded = true;
7561 } else {
7562 teseInBuiltins.remove(builtin.type);
7563 continue;
7564 }
7565 } else {
7566 variable.type = QShaderDescription::VariableType::Half2;
7567 binding = tessFactorBufferBinding;
7568 offsetInTessFactorBuffer = 8;
7569 offset = &offsetInTessFactorBuffer;
7570 alignment = &tessFactorAlignment;
7571 }
7572 break;
7573 default:
7574 Q_UNREACHABLE();
7575 break;
7576 }
7577
7578 if (teseInBuiltins.contains(builtin.type)) {
7579 if (binding != -1) {
7580 int index = nextAttributeIndex(indices);
7581 addVertexAttribute(variable, binding, rhiD, index, *offset, vertexDesc.attributes, indices, *alignment);
7582 usedBuffers << binding;
7583 } else {
7584 qWarning() << "baked tessellation control shader missing output buffer binding information";
7585 addUnusedVertexAttribute(variable, rhiD, *offset, *alignment);
7586 }
7587 } else {
7588 addUnusedVertexAttribute(variable, rhiD, *offset, *alignment);
7589 }
7590
7591 teseInBuiltins.remove(builtin.type);
7592 }
7593
7594 for (const QShaderDescription::BuiltinVariable &builtin : teseInBuiltins) {
7595 switch (builtin.type) {
7596 case QShaderDescription::BuiltinType::PositionBuiltin:
7597 case QShaderDescription::BuiltinType::PointSizeBuiltin:
7598 case QShaderDescription::BuiltinType::ClipDistanceBuiltin:
7599 qWarning() << "missing tessellation control output for tessellation evaluation builtin input:" << builtin;
7600 break;
7601 default:
7602 break;
7603 }
7604 }
7605
7606 if (usedBuffers.contains(tescOutputBufferBinding)) {
7607 vertexDesc.layouts[tescOutputBufferBinding].stepFunction = MTLVertexStepFunctionPerPatchControlPoint;
7608 vertexDesc.layouts[tescOutputBufferBinding].stride = aligned(offsetInTescOutput, tescOutputAlignment);
7609 }
7610
7611 if (usedBuffers.contains(tescPatchOutputBufferBinding)) {
7612 vertexDesc.layouts[tescPatchOutputBufferBinding].stepFunction = MTLVertexStepFunctionPerPatch;
7613 vertexDesc.layouts[tescPatchOutputBufferBinding].stride = aligned(offsetInTescPatchOutput, tescPatchOutputAlignment);
7614 }
7615
7616 if (usedBuffers.contains(tessFactorBufferBinding)) {
7617 vertexDesc.layouts[tessFactorBufferBinding].stepFunction = MTLVertexStepFunctionPerPatch;
7618 vertexDesc.layouts[tessFactorBufferBinding].stride = trianglesMode ? sizeof(MTLTriangleTessellationFactorsHalf) : sizeof(MTLQuadTessellationFactorsHalf);
7619 }
7620
7621 rpDesc.vertexDescriptor = vertexDesc;
7622 rpDesc.vertexFunction = vertTese.func;
7623 rpDesc.fragmentFunction = pipeline->d->fs.func;
7624
7625 // The portable, cross-API approach is to use CCW, the results are then
7626 // identical (assuming the applied clipSpaceCorrMatrix) for all the 3D
7627 // APIs. The tess.eval. GLSL shader is thus expected to specify ccw. If it
7628 // doesn't, things may not work as expected.
7629 rpDesc.tessellationOutputWindingOrder = toMetalTessellationWindingOrder(vertTese.desc.tessellationWindingOrder());
7630
7631 rpDesc.tessellationPartitionMode = toMetalTessellationPartitionMode(vertTese.desc.tessellationPartitioning());
7632
7633 QMetalRenderPassDescriptor *rpD = QRHI_RES(QMetalRenderPassDescriptor, pipeline->renderPassDescriptor());
7635
7636 rhiD->d->trySeedingRenderPipelineFromBinaryArchive(rpDesc);
7637
7638 if (rhiD->rhiFlags.testFlag(QRhi::EnablePipelineCacheDataSave))
7639 rhiD->d->addRenderPipelineToBinaryArchive(rpDesc);
7640
7641 NSError *err = nil;
7642 id<MTLRenderPipelineState> ps = [rhiD->d->dev newRenderPipelineStateWithDescriptor: rpDesc error: &err];
7643 [rpDesc release];
7644 if (!ps) {
7645 const QString msg = QString::fromNSString(err.localizedDescription);
7646 qWarning("Failed to create render pipeline state for tessellation: %s", qPrintable(msg));
7647 } else {
7648 // ps is stored in the QMetalGraphicsPipelineData so the end result in this
7649 // regard is no different from what createVertexFragmentPipeline does
7650 pipeline->d->ps = ps;
7651 }
7652 return ps;
7653}
7654
7656{
7657 QVector<QMetalBuffer *> *workBuffers = type == WorkBufType::DeviceLocal ? &deviceLocalWorkBuffers : &hostVisibleWorkBuffers;
7658
7659 // Check if something is reusable as-is.
7660 for (QMetalBuffer *workBuf : *workBuffers) {
7661 if (workBuf && workBuf->lastActiveFrameSlot == -1 && workBuf->size() >= size) {
7662 workBuf->lastActiveFrameSlot = rhiD->currentFrameSlot;
7663 return workBuf;
7664 }
7665 }
7666
7667 // Once the pool is above a certain threshold, see if there is something
7668 // unused (but too small) and recreate that our size.
7669 if (workBuffers->count() > QMTL_FRAMES_IN_FLIGHT * 8) {
7670 for (QMetalBuffer *workBuf : *workBuffers) {
7671 if (workBuf && workBuf->lastActiveFrameSlot == -1) {
7672 workBuf->setSize(size);
7673 if (workBuf->create()) {
7674 workBuf->lastActiveFrameSlot = rhiD->currentFrameSlot;
7675 return workBuf;
7676 }
7677 }
7678 }
7679 }
7680
7681 // Add a new buffer to the pool.
7682 QMetalBuffer *buf;
7683 if (type == WorkBufType::DeviceLocal) {
7684 // for GPU->GPU data (non-slotted, not necessarily host writable)
7685 buf = new QMetalBuffer(rhiD, QRhiBuffer::Static, QRhiBuffer::UsageFlags(QMetalBuffer::WorkBufPoolUsage), size);
7686 } else {
7687 // for CPU->GPU (non-slotted, host writable/coherent)
7688 buf = new QMetalBuffer(rhiD, QRhiBuffer::Dynamic, QRhiBuffer::UsageFlags(QMetalBuffer::WorkBufPoolUsage), size);
7689 }
7690 if (buf->create()) {
7691 buf->lastActiveFrameSlot = rhiD->currentFrameSlot;
7692 workBuffers->append(buf);
7693 return buf;
7694 }
7695
7696 qWarning("Failed to acquire work buffer of size %u", size);
7697 return nullptr;
7698}
7699
7700bool QMetalGraphicsPipeline::createTessellationPipelines(const QShader &tessVert, const QShader &tesc, const QShader &tese, const QShader &tessFrag)
7701{
7702 QRHI_RES_RHI(QRhiMetal);
7703 QString error;
7704 QByteArray entryPoint;
7705 QShaderKey activeKey;
7706
7707 const QShaderDescription tescDesc = tesc.description();
7708 const QShaderDescription teseDesc = tese.description();
7709 d->tess.inControlPointCount = uint(m_patchControlPointCount);
7710 d->tess.outControlPointCount = tescDesc.tessellationOutputVertexCount();
7711 if (!d->tess.outControlPointCount)
7712 d->tess.outControlPointCount = teseDesc.tessellationOutputVertexCount();
7713
7714 if (!d->tess.outControlPointCount) {
7715 qWarning("Failed to determine output vertex count from the tessellation control or evaluation shader, cannot tessellate");
7716 d->tess.enabled = false;
7717 d->tess.failed = true;
7718 return false;
7719 }
7720
7721 if (m_multiViewCount >= 2)
7722 qWarning("Multiview is not supported with tessellation");
7723
7724 // Now the vertex shader is a compute shader.
7725 // It should have three dedicated *VertexAsComputeShader variants.
7726 // What the requested variant was (Standard or Batchable) plays no role here.
7727 // (the Qt Quick scenegraph does not use tessellation with its materials)
7728 // Create all three versions.
7729
7730 bool variantsPresent[3] = {};
7731 const QVector<QShaderKey> tessVertKeys = tessVert.availableShaders();
7732 for (const QShaderKey &k : tessVertKeys) {
7733 switch (k.sourceVariant()) {
7734 case QShader::NonIndexedVertexAsComputeShader:
7735 variantsPresent[0] = true;
7736 break;
7737 case QShader::UInt32IndexedVertexAsComputeShader:
7738 variantsPresent[1] = true;
7739 break;
7740 case QShader::UInt16IndexedVertexAsComputeShader:
7741 variantsPresent[2] = true;
7742 break;
7743 default:
7744 break;
7745 }
7746 }
7747 if (!(variantsPresent[0] && variantsPresent[1] && variantsPresent[2])) {
7748 qWarning("Vertex shader is not prepared for Metal tessellation. Cannot tessellate. "
7749 "Perhaps the relevant variants (UInt32IndexedVertexAsComputeShader et al) were not generated? "
7750 "Try passing --msltess to qsb.");
7751 d->tess.enabled = false;
7752 d->tess.failed = true;
7753 return false;
7754 }
7755
7756 int varIndex = 0; // Will map NonIndexed as 0, UInt32 as 1, UInt16 as 2. Do not change this ordering.
7757 for (QShader::Variant variant : {
7758 QShader::NonIndexedVertexAsComputeShader,
7759 QShader::UInt32IndexedVertexAsComputeShader,
7760 QShader::UInt16IndexedVertexAsComputeShader })
7761 {
7762 id<MTLLibrary> lib = rhiD->d->createMetalLib(tessVert, variant, false, &error, &entryPoint, &activeKey);
7763 if (!lib) {
7764 qWarning("MSL shader compilation failed for vertex-as-compute shader %d: %s", int(variant), qPrintable(error));
7765 d->tess.enabled = false;
7766 d->tess.failed = true;
7767 return false;
7768 }
7769 id<MTLFunction> func = rhiD->d->createMSLShaderFunction(lib, entryPoint);
7770 if (!func) {
7771 qWarning("MSL function for entry point %s not found", entryPoint.constData());
7772 [lib release];
7773 d->tess.enabled = false;
7774 d->tess.failed = true;
7775 return false;
7776 }
7777 QMetalShader &compVs(d->tess.compVs[varIndex]);
7778 compVs.lib = lib;
7779 compVs.func = func;
7780 compVs.desc = tessVert.description();
7781 compVs.nativeResourceBindingMap = tessVert.nativeResourceBindingMap(activeKey);
7782 compVs.nativeShaderInfo = tessVert.nativeShaderInfo(activeKey);
7783
7784 // pre-create all three MTLComputePipelineStates
7785 if (!d->tess.vsCompPipeline(rhiD, variant)) {
7786 qWarning("Failed to pre-generate compute pipeline for vertex compute shader (tessellation variant %d)", int(variant));
7787 d->tess.enabled = false;
7788 d->tess.failed = true;
7789 return false;
7790 }
7791
7792 ++varIndex;
7793 }
7794
7795 // Pipeline #2 is a compute that runs the tessellation control (compute) shader
7796 id<MTLLibrary> tessControlLib = rhiD->d->createMetalLib(tesc, QShader::StandardShader, false, &error, &entryPoint, &activeKey);
7797 if (!tessControlLib) {
7798 qWarning("MSL shader compilation failed for tessellation control compute shader: %s", qPrintable(error));
7799 d->tess.enabled = false;
7800 d->tess.failed = true;
7801 return false;
7802 }
7803 id<MTLFunction> tessControlFunc = rhiD->d->createMSLShaderFunction(tessControlLib, entryPoint);
7804 if (!tessControlFunc) {
7805 qWarning("MSL function for entry point %s not found", entryPoint.constData());
7806 [tessControlLib release];
7807 d->tess.enabled = false;
7808 d->tess.failed = true;
7809 return false;
7810 }
7811 d->tess.compTesc.lib = tessControlLib;
7812 d->tess.compTesc.func = tessControlFunc;
7813 d->tess.compTesc.desc = tesc.description();
7814 d->tess.compTesc.nativeResourceBindingMap = tesc.nativeResourceBindingMap(activeKey);
7815 d->tess.compTesc.nativeShaderInfo = tesc.nativeShaderInfo(activeKey);
7816 if (!d->tess.tescCompPipeline(rhiD)) {
7817 qWarning("Failed to pre-generate compute pipeline for tessellation control shader");
7818 d->tess.enabled = false;
7819 d->tess.failed = true;
7820 return false;
7821 }
7822
7823 // Pipeline #3 is a render pipeline with the tessellation evaluation (vertex) + the fragment shader
7824 id<MTLLibrary> tessEvalLib = rhiD->d->createMetalLib(tese, QShader::StandardShader, false, &error, &entryPoint, &activeKey);
7825 if (!tessEvalLib) {
7826 qWarning("MSL shader compilation failed for tessellation evaluation vertex shader: %s", qPrintable(error));
7827 d->tess.enabled = false;
7828 d->tess.failed = true;
7829 return false;
7830 }
7831 id<MTLFunction> tessEvalFunc = rhiD->d->createMSLShaderFunction(tessEvalLib, entryPoint);
7832 if (!tessEvalFunc) {
7833 qWarning("MSL function for entry point %s not found", entryPoint.constData());
7834 [tessEvalLib release];
7835 d->tess.enabled = false;
7836 d->tess.failed = true;
7837 return false;
7838 }
7839 d->tess.vertTese.lib = tessEvalLib;
7840 d->tess.vertTese.func = tessEvalFunc;
7841 d->tess.vertTese.desc = tese.description();
7842 d->tess.vertTese.nativeResourceBindingMap = tese.nativeResourceBindingMap(activeKey);
7843 d->tess.vertTese.nativeShaderInfo = tese.nativeShaderInfo(activeKey);
7844
7845 id<MTLLibrary> fragLib = rhiD->d->createMetalLib(tessFrag, QShader::StandardShader, false, &error, &entryPoint, &activeKey);
7846 if (!fragLib) {
7847 qWarning("MSL shader compilation failed for fragment shader: %s", qPrintable(error));
7848 d->tess.enabled = false;
7849 d->tess.failed = true;
7850 return false;
7851 }
7852 id<MTLFunction> fragFunc = rhiD->d->createMSLShaderFunction(fragLib, entryPoint);
7853 if (!fragFunc) {
7854 qWarning("MSL function for entry point %s not found", entryPoint.constData());
7855 [fragLib release];
7856 d->tess.enabled = false;
7857 d->tess.failed = true;
7858 return false;
7859 }
7860 d->fs.lib = fragLib;
7861 d->fs.func = fragFunc;
7862 d->fs.desc = tessFrag.description();
7863 d->fs.nativeShaderInfo = tessFrag.nativeShaderInfo(activeKey);
7864 d->fs.nativeResourceBindingMap = tessFrag.nativeResourceBindingMap(activeKey);
7865
7866 if (!d->tess.teseFragRenderPipeline(rhiD, this)) {
7867 qWarning("Failed to pre-generate render pipeline for tessellation evaluation + fragment shader");
7868 d->tess.enabled = false;
7869 d->tess.failed = true;
7870 return false;
7871 }
7872
7873 MTLDepthStencilDescriptor *dsDesc = [[MTLDepthStencilDescriptor alloc] init];
7875 d->ds = [rhiD->d->dev newDepthStencilStateWithDescriptor: dsDesc];
7876 [dsDesc release];
7877
7878 // no primitiveType
7880
7881 return true;
7882}
7883
7885{
7886 destroy(); // no early test, always invoke and leave it to destroy to decide what to clean up
7887
7888 QRHI_RES_RHI(QRhiMetal);
7889 rhiD->pipelineCreationStart();
7890 if (!rhiD->sanityCheckGraphicsPipeline(this))
7891 return false;
7892
7893 // See if tessellation is involved. Things will be very different, if so.
7894 QShader tessVert;
7895 QShader tesc;
7896 QShader tese;
7897 QShader tessFrag;
7898 for (const QRhiShaderStage &shaderStage : std::as_const(m_shaderStages)) {
7899 switch (shaderStage.type()) {
7900 case QRhiShaderStage::Vertex:
7901 tessVert = shaderStage.shader();
7902 break;
7903 case QRhiShaderStage::TessellationControl:
7904 tesc = shaderStage.shader();
7905 break;
7906 case QRhiShaderStage::TessellationEvaluation:
7907 tese = shaderStage.shader();
7908 break;
7909 case QRhiShaderStage::Fragment:
7910 tessFrag = shaderStage.shader();
7911 break;
7912 default:
7913 break;
7914 }
7915 }
7916 d->tess.enabled = tesc.isValid() && tese.isValid() && m_topology == Patches && m_patchControlPointCount > 0;
7917 d->tess.failed = false;
7918
7919 bool ok = d->tess.enabled ? createTessellationPipelines(tessVert, tesc, tese, tessFrag) : createVertexFragmentPipeline();
7920 if (!ok)
7921 return false;
7922
7923 // SPIRV-Cross buffer size buffers
7924 int buffers = 0;
7925 QVarLengthArray<QMetalShader *, 6> shaders;
7926 if (d->tess.enabled) {
7927 shaders.append(&d->tess.compVs[0]);
7928 shaders.append(&d->tess.compVs[1]);
7929 shaders.append(&d->tess.compVs[2]);
7930 shaders.append(&d->tess.compTesc);
7931 shaders.append(&d->tess.vertTese);
7932 } else {
7933 shaders.append(&d->vs);
7934 }
7935 shaders.append(&d->fs);
7936
7937 for (QMetalShader *shader : shaders) {
7938 if (shader->nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding)) {
7939 const int binding = shader->nativeShaderInfo.extraBufferBindings[QShaderPrivate::MslBufferSizeBufferBinding];
7940 shader->nativeResourceBindingMap[binding] = {binding, -1};
7941 int maxNativeBinding = 0;
7942 for (const QShaderDescription::StorageBlock &block : shader->desc.storageBlocks())
7943 maxNativeBinding = qMax(maxNativeBinding, shader->nativeResourceBindingMap[block.binding].first);
7944
7945 // we use one buffer to hold data for all graphics shader stages, each with a different offset.
7946 // buffer offsets must be 32byte aligned - adjust buffer count accordingly
7947 buffers += ((maxNativeBinding + 1 + 7) / 8) * 8;
7948 }
7949 }
7950
7951 if (buffers) {
7952 if (!d->bufferSizeBuffer)
7953 d->bufferSizeBuffer = new QMetalBuffer(rhiD, QRhiBuffer::Static,
7954 QRhiBuffer::UsageFlags(int(QRhiBuffer::StorageBuffer)
7955 | QMetalBuffer::InternalHostWritable),
7956 buffers * sizeof(int));
7957
7958 d->bufferSizeBuffer->setSize(buffers * sizeof(int));
7960 }
7961
7962 rhiD->pipelineCreationEnd();
7964 generation += 1;
7965 rhiD->registerResource(this);
7966 return true;
7967}
7968
7974
7976{
7977 destroy();
7978 delete d;
7979}
7980
7982{
7983 d->cs.destroy();
7984
7985 if (!d->ps)
7986 return;
7987
7988 delete d->bufferSizeBuffer;
7989 d->bufferSizeBuffer = nullptr;
7990
7994 e.computePipeline.pipelineState = d->ps;
7995 d->ps = nil;
7996
7997 QRHI_RES_RHI(QRhiMetal);
7998 if (rhiD) {
7999 rhiD->d->releaseQueue.append(e);
8000 rhiD->unregisterResource(this);
8001 }
8002}
8003
8004void QRhiMetalData::trySeedingComputePipelineFromBinaryArchive(MTLComputePipelineDescriptor *cpDesc)
8005{
8006 if (binArch) {
8007 NSArray *binArchArray = [NSArray arrayWithObjects: binArch, nil];
8008 cpDesc.binaryArchives = binArchArray;
8009 }
8010}
8011
8012void QRhiMetalData::addComputePipelineToBinaryArchive(MTLComputePipelineDescriptor *cpDesc)
8013{
8014 if (binArch) {
8015 NSError *err = nil;
8016 if (![binArch addComputePipelineFunctionsWithDescriptor: cpDesc error: &err]) {
8017 const QString msg = QString::fromNSString(err.localizedDescription);
8018 qWarning("Failed to collect compute pipeline functions to binary archive: %s", qPrintable(msg));
8019 }
8020 }
8021}
8022
8024{
8025 if (d->ps)
8026 destroy();
8027
8028 QRHI_RES_RHI(QRhiMetal);
8029 rhiD->pipelineCreationStart();
8030
8031 auto cacheIt = rhiD->d->shaderCache.constFind({ m_shaderStage, false });
8032 if (cacheIt != rhiD->d->shaderCache.constEnd()) {
8033 d->cs = *cacheIt;
8034 } else {
8035 const QShader shader = m_shaderStage.shader();
8036 QString error;
8037 QByteArray entryPoint;
8038 QShaderKey activeKey;
8039 id<MTLLibrary> lib = rhiD->d->createMetalLib(shader, m_shaderStage.shaderVariant(), false,
8040 &error, &entryPoint, &activeKey);
8041 if (!lib) {
8042 qWarning("MSL shader compilation failed: %s", qPrintable(error));
8043 return false;
8044 }
8045 id<MTLFunction> func = rhiD->d->createMSLShaderFunction(lib, entryPoint);
8046 if (!func) {
8047 qWarning("MSL function for entry point %s not found", entryPoint.constData());
8048 [lib release];
8049 return false;
8050 }
8051 d->cs.lib = lib;
8052 d->cs.func = func;
8053 d->cs.localSize = shader.description().computeShaderLocalSize();
8054 d->cs.nativeResourceBindingMap = shader.nativeResourceBindingMap(activeKey);
8055 d->cs.desc = shader.description();
8056 d->cs.nativeShaderInfo = shader.nativeShaderInfo(activeKey);
8057
8058 // Compute never needs argument buffers on its own (they are there for
8059 // indirect command buffers, which are graphics-only), hence not asking
8060 // for that shader variant above. It can still be requested explicitly
8061 // via the QRhiShaderStage, in which case the textures and samplers have
8062 // to go through an argument buffer here as well.
8063 setupArgumentBufferEncoder(&d->cs);
8064
8065 // SPIRV-Cross buffer size buffers
8066 if (d->cs.nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding)) {
8067 const int binding = d->cs.nativeShaderInfo.extraBufferBindings[QShaderPrivate::MslBufferSizeBufferBinding];
8068 d->cs.nativeResourceBindingMap[binding] = {binding, -1};
8069 }
8070
8071 if (rhiD->d->shaderCache.count() >= QRhiMetal::MAX_SHADER_CACHE_ENTRIES) {
8072 for (QMetalShader &s : rhiD->d->shaderCache)
8073 s.destroy();
8074 rhiD->d->shaderCache.clear();
8075 }
8076 rhiD->d->shaderCache.insert({ m_shaderStage, false }, d->cs);
8077 }
8078
8079 [d->cs.lib retain];
8080 [d->cs.func retain];
8081 [d->cs.argumentEncoder retain];
8082
8083 if (d->cs.argumentBufferIndex >= 0 && !rhiD->caps.indirectCommandBuffers) {
8084 // Sampler states are created with supportArgumentBuffers only when this
8085 // cap is present, and Metal faults the GPU without any warning when one
8086 // that lacks it is used with an argument buffer.
8087 qWarning("The ArgumentBufferShader variant of a compute shader cannot be used on this device");
8088 return false;
8089 }
8090
8091 d->localSize = MTLSizeMake(d->cs.localSize[0], d->cs.localSize[1], d->cs.localSize[2]);
8092
8093 MTLComputePipelineDescriptor *cpDesc = [MTLComputePipelineDescriptor new];
8094 cpDesc.computeFunction = d->cs.func;
8095
8096 rhiD->d->trySeedingComputePipelineFromBinaryArchive(cpDesc);
8097
8098 if (rhiD->rhiFlags.testFlag(QRhi::EnablePipelineCacheDataSave))
8099 rhiD->d->addComputePipelineToBinaryArchive(cpDesc);
8100
8101 NSError *err = nil;
8102 d->ps = [rhiD->d->dev newComputePipelineStateWithDescriptor: cpDesc
8103 options: MTLPipelineOptionNone
8104 reflection: nil
8105 error: &err];
8106 [cpDesc release];
8107 if (!d->ps) {
8108 const QString msg = QString::fromNSString(err.localizedDescription);
8109 qWarning("Failed to create compute pipeline state: %s", qPrintable(msg));
8110 return false;
8111 }
8112
8113 // SPIRV-Cross buffer size buffers
8114 if (d->cs.nativeShaderInfo.extraBufferBindings.contains(QShaderPrivate::MslBufferSizeBufferBinding)) {
8115 int buffers = 0;
8116 for (const QShaderDescription::StorageBlock &block : d->cs.desc.storageBlocks())
8117 buffers = qMax(buffers, d->cs.nativeResourceBindingMap[block.binding].first);
8118
8119 buffers += 1;
8120
8121 if (!d->bufferSizeBuffer)
8122 d->bufferSizeBuffer = new QMetalBuffer(rhiD, QRhiBuffer::Static,
8123 QRhiBuffer::UsageFlags(int(QRhiBuffer::StorageBuffer)
8124 | QMetalBuffer::InternalHostWritable),
8125 buffers * sizeof(int));
8126
8127 d->bufferSizeBuffer->setSize(buffers * sizeof(int));
8129 }
8130
8131 rhiD->pipelineCreationEnd();
8133 generation += 1;
8134 rhiD->registerResource(this);
8135 return true;
8136}
8137
8141{
8143}
8144
8146{
8147 destroy();
8148 delete d;
8149}
8150
8152{
8153 // nothing to do here, we do not own the MTL cb object
8154}
8155
8157{
8158 nativeHandlesStruct.commandBuffer = (MTLCommandBuffer *) d->cb;
8159 nativeHandlesStruct.encoder = (MTLRenderCommandEncoder *) d->currentRenderPassEncoder;
8160 return &nativeHandlesStruct;
8161}
8162
8163void QMetalCommandBuffer::resetState(double lastGpuTime)
8164{
8165 d->lastGpuTime = lastGpuTime;
8166 d->currentRenderPassEncoder = nil;
8167 d->currentComputePassEncoder = nil;
8168 d->tessellationComputeEncoder = nil;
8169 d->currentPassRpDesc = nil;
8171}
8172
8174{
8176 currentTarget = nullptr;
8177 d->openDebugGroups.clear();
8179}
8180
8182{
8183 currentGraphicsPipeline = nullptr;
8184 currentComputePipeline = nullptr;
8186 pushConstantData.clear();
8188 currentGraphicsSrb = nullptr;
8189 currentComputeSrb = nullptr;
8191 currentResSlot = -1;
8192 currentIndexBuffer = nullptr;
8193 currentIndexOffset = 0;
8194 currentIndexFormat = QRhiCommandBuffer::IndexUInt16;
8195 currentCullMode = -1;
8199 currentDepthBiasValues = { 0.0f, 0.0f };
8200 hasCustomScissorSet = false;
8201 currentScissor = {};
8202 currentViewport = {};
8203 hasBlendConstantsSet = false;
8204 currentBlendConstants = {};
8205 hasStencilRefSet = false;
8206 currentStencilRef = 0;
8207
8208 d->currentShaderResourceBindingState = {};
8209 d->currentDepthStencilState = nil;
8211 d->currentVertexInputsBuffers.clear();
8212 d->currentVertexInputOffsets.clear();
8213}
8214
8215QMetalSwapChain::QMetalSwapChain(QRhiImplementation *rhi)
8216 : QRhiSwapChain(rhi),
8217 rtWrapper(rhi, this),
8218 cbWrapper(rhi),
8220{
8221 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
8222 d->sem[i] = nullptr;
8223 d->msaaTex[i] = nil;
8224 }
8225}
8226
8228{
8229 destroy();
8230 delete d;
8231}
8232
8234{
8235 if (!d->layer)
8236 return;
8237
8238 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
8239 if (d->sem[i]) {
8240 // the semaphores cannot be released if they do not have the initial value
8242
8243 dispatch_release(d->sem[i]);
8244 d->sem[i] = nullptr;
8245 }
8246 }
8247
8248 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
8249 [d->msaaTex[i] release];
8250 d->msaaTex[i] = nil;
8251 }
8252
8253 d->layer = nullptr;
8254 m_proxyData = {};
8255
8256 [d->curDrawable release];
8257 d->curDrawable = nil;
8258
8259 QRHI_RES_RHI(QRhiMetal);
8260 if (rhiD) {
8261 rhiD->swapchains.remove(this);
8262 rhiD->unregisterResource(this);
8263 }
8264}
8265
8267{
8268 return &cbWrapper;
8269}
8270
8275
8276// view.layer should ideally be called on the main thread, otherwise the UI
8277// Thread Checker in Xcode drops a warning. Hence trying to proxy it through
8278// QRhiSwapChainProxyData instead of just calling this function directly.
8279static inline CAMetalLayer *layerForWindow(QWindow *window)
8280{
8281 Q_ASSERT(window);
8282 CALayer *layer = nullptr;
8283#ifdef Q_OS_MACOS
8284 if (auto *cocoaWindow = window->nativeInterface<QNativeInterface::Private::QCocoaWindow>())
8285 layer = cocoaWindow->contentLayer();
8286#else
8287 layer = reinterpret_cast<UIView *>(window->winId()).layer;
8288#endif
8289 Q_ASSERT(layer);
8290 return static_cast<CAMetalLayer *>(layer);
8291}
8292
8293// If someone calls this, it is hopefully from the main thread, and they will
8294// then set the returned data on the QRhiSwapChain, so it won't need to query
8295// the layer on its own later on.
8297{
8299 d.reserved[0] = layerForWindow(window);
8300 return d;
8301}
8302
8304{
8305 Q_ASSERT(m_window);
8306 CAMetalLayer *layer = d->layer;
8307 if (!layer)
8308 layer = qrhi_objectFromProxyData<CAMetalLayer>(&m_proxyData, m_window, QRhi::Metal, 0);
8309
8310 Q_ASSERT(layer);
8311 int height = (int)layer.bounds.size.height;
8312 int width = (int)layer.bounds.size.width;
8313 width *= layer.contentsScale;
8314 height *= layer.contentsScale;
8315 return QSize(width, height);
8316}
8317
8319{
8320 if (f == HDRExtendedSrgbLinear) {
8321 return hdrInfo().limits.colorComponentValue.maxPotentialColorComponentValue > 1.0f;
8322 } else if (f == HDR10) {
8323 return hdrInfo().limits.colorComponentValue.maxPotentialColorComponentValue > 1.0f;
8324 } else if (f == HDRExtendedDisplayP3Linear) {
8325 return hdrInfo().limits.colorComponentValue.maxPotentialColorComponentValue > 1.0f;
8326 }
8327 return f == SDR;
8328}
8329
8331{
8332 QRHI_RES_RHI(QRhiMetal);
8333
8334 chooseFormats(); // ensure colorFormat and similar are filled out
8335
8336 QMetalRenderPassDescriptor *rpD = new QMetalRenderPassDescriptor(m_rhi);
8338 rpD->hasDepthStencil = m_depthStencil != nullptr;
8339
8340 rpD->colorFormat[0] = int(d->colorFormat);
8341
8342#ifdef Q_OS_MACOS
8343 // m_depthStencil may not be built yet so cannot rely on computed fields in it
8344 rpD->dsFormat = rhiD->d->dev.depth24Stencil8PixelFormatSupported
8345 ? MTLPixelFormatDepth24Unorm_Stencil8 : MTLPixelFormatDepth32Float_Stencil8;
8346#else
8347 rpD->dsFormat = MTLPixelFormatDepth32Float_Stencil8;
8348#endif
8349
8350 rpD->hasShadingRateMap = m_shadingRateMap != nullptr;
8351
8353
8354 rhiD->registerResource(rpD, false);
8355 return rpD;
8356}
8357
8359{
8360 QRHI_RES_RHI(QRhiMetal);
8361 samples = rhiD->effectiveSampleCount(m_sampleCount);
8362 // pick a format that is allowed for CAMetalLayer.pixelFormat
8363 if (m_format == HDRExtendedSrgbLinear || m_format == HDRExtendedDisplayP3Linear) {
8364 d->colorFormat = MTLPixelFormatRGBA16Float;
8365 d->rhiColorFormat = QRhiTexture::RGBA16F;
8366 return;
8367 }
8368 if (m_format == HDR10) {
8369 d->colorFormat = MTLPixelFormatRGB10A2Unorm;
8370 d->rhiColorFormat = QRhiTexture::RGB10A2;
8371 return;
8372 }
8373 d->colorFormat = m_flags.testFlag(sRGB) ? MTLPixelFormatBGRA8Unorm_sRGB : MTLPixelFormatBGRA8Unorm;
8374 d->rhiColorFormat = QRhiTexture::BGRA8;
8375}
8376
8378{
8379 // wait+signal is the general pattern to ensure the commands for a
8380 // given frame slot have completed (if sem is 1, we go 0 then 1; if
8381 // sem is 0 we go -1, block, completion increments to 0, then us to 1)
8382
8383 dispatch_semaphore_t sem = d->sem[slot];
8384 dispatch_semaphore_wait(sem, DISPATCH_TIME_FOREVER);
8385 dispatch_semaphore_signal(sem);
8386}
8387
8389{
8390 Q_ASSERT(m_window);
8391
8392 const bool needsRegistration = !window || window != m_window;
8393
8394 if (window && window != m_window)
8395 destroy();
8396 // else no destroy(), this is intentional
8397
8398 QRHI_RES_RHI(QRhiMetal);
8399 if (needsRegistration || !rhiD->swapchains.contains(this))
8400 rhiD->swapchains.insert(this);
8401
8402 window = m_window;
8403
8404 if (window->surfaceType() != QSurface::MetalSurface) {
8405 qWarning("QMetalSwapChain only supports MetalSurface windows");
8406 return false;
8407 }
8408
8409 d->layer = qrhi_objectFromProxyData<CAMetalLayer>(&m_proxyData, window, QRhi::Metal, 0);
8410 Q_ASSERT(d->layer);
8411
8413 if (d->colorFormat != d->layer.pixelFormat)
8414 d->layer.pixelFormat = d->colorFormat;
8415
8416 if (m_format == HDRExtendedSrgbLinear) {
8417 d->layer.colorspace = CGColorSpaceCreateWithName(kCGColorSpaceExtendedLinearSRGB);
8418 d->layer.wantsExtendedDynamicRangeContent = YES;
8419 } else if (m_format == HDR10) {
8420 d->layer.colorspace = CGColorSpaceCreateWithName(kCGColorSpaceITUR_2100_PQ);
8421 d->layer.wantsExtendedDynamicRangeContent = YES;
8422 } else if (m_format == HDRExtendedDisplayP3Linear) {
8423 d->layer.colorspace = CGColorSpaceCreateWithName(kCGColorSpaceExtendedLinearDisplayP3);
8424 d->layer.wantsExtendedDynamicRangeContent = YES;
8425 }
8426
8427 if (m_flags.testFlag(UsedAsTransferSource))
8428 d->layer.framebufferOnly = NO;
8429
8430#ifdef Q_OS_MACOS
8431 if (m_flags.testFlag(NoVSync))
8432 d->layer.displaySyncEnabled = NO;
8433#endif
8434
8435 if (m_flags.testFlag(SurfaceHasPreMulAlpha)) {
8436 d->layer.opaque = NO;
8437 } else if (m_flags.testFlag(SurfaceHasNonPreMulAlpha)) {
8438 // The CoreAnimation compositor is said to expect premultiplied alpha,
8439 // so this is then wrong when it comes to the blending operations but
8440 // there's nothing we can do. Fortunately Qt Quick always outputs
8441 // premultiplied alpha so it is not a problem there.
8442 d->layer.opaque = NO;
8443 } else {
8444 d->layer.opaque = YES;
8445 }
8446
8447 // Now set the layer's drawableSize which will stay set to the same value
8448 // until the next createOrResize(), thus ensuring atomicity with regards to
8449 // the drawable size in frames.
8450 int width = (int)d->layer.bounds.size.width;
8451 int height = (int)d->layer.bounds.size.height;
8452 CGSize layerSize = CGSizeMake(width, height);
8453 const float scaleFactor = d->layer.contentsScale;
8454 layerSize.width *= scaleFactor;
8455 layerSize.height *= scaleFactor;
8456 d->layer.drawableSize = layerSize;
8457
8458 m_currentPixelSize = QSizeF::fromCGSize(layerSize).toSize();
8459 pixelSize = m_currentPixelSize;
8460
8461 [d->layer setDevice: rhiD->d->dev];
8462
8463 [d->curDrawable release];
8464 d->curDrawable = nil;
8465
8466 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
8467 d->lastGpuTime[i] = 0;
8468 if (!d->sem[i])
8469 d->sem[i] = dispatch_semaphore_create(QMTL_FRAMES_IN_FLIGHT - 1);
8470 }
8471
8472 currentFrameSlot = 0;
8473 frameCount = 0;
8474
8475 ds = m_depthStencil ? QRHI_RES(QMetalRenderBuffer, m_depthStencil) : nullptr;
8476 if (m_depthStencil && m_depthStencil->sampleCount() != m_sampleCount) {
8477 qWarning("Depth-stencil buffer's sampleCount (%d) does not match color buffers' sample count (%d). Expect problems.",
8478 m_depthStencil->sampleCount(), m_sampleCount);
8479 }
8480 if (m_depthStencil && m_depthStencil->pixelSize() != pixelSize) {
8481 if (m_depthStencil->flags().testFlag(QRhiRenderBuffer::UsedWithSwapChainOnly)) {
8482 m_depthStencil->setPixelSize(pixelSize);
8483 if (!m_depthStencil->create())
8484 qWarning("Failed to rebuild swapchain's associated depth-stencil buffer for size %dx%d",
8485 pixelSize.width(), pixelSize.height());
8486 } else {
8487 qWarning("Depth-stencil buffer's size (%dx%d) does not match the layer size (%dx%d). Expect problems.",
8488 m_depthStencil->pixelSize().width(), m_depthStencil->pixelSize().height(),
8489 pixelSize.width(), pixelSize.height());
8490 }
8491 }
8492
8493 rtWrapper.setRenderPassDescriptor(m_renderPassDesc); // for the public getter in QRhiRenderTarget
8494 rtWrapper.d->pixelSize = pixelSize;
8495 rtWrapper.d->dpr = scaleFactor;
8498 rtWrapper.d->dsAttCount = ds ? 1 : 0;
8499
8500 qCDebug(QRHI_LOG_INFO, "got CAMetalLayer, pixel size %dx%d (scale %.2f)",
8501 pixelSize.width(), pixelSize.height(), scaleFactor);
8502
8503 if (samples > 1) {
8504 MTLTextureDescriptor *desc = [[MTLTextureDescriptor alloc] init];
8505 desc.textureType = MTLTextureType2DMultisample;
8506 desc.pixelFormat = d->colorFormat;
8507 desc.width = NSUInteger(pixelSize.width());
8508 desc.height = NSUInteger(pixelSize.height());
8509 desc.sampleCount = NSUInteger(samples);
8510 desc.resourceOptions = MTLResourceStorageModePrivate;
8511 desc.storageMode = MTLStorageModePrivate;
8512 desc.usage = MTLTextureUsageRenderTarget;
8513 for (int i = 0; i < QMTL_FRAMES_IN_FLIGHT; ++i) {
8514 if (d->msaaTex[i]) {
8517 e.lastActiveFrameSlot = 1; // because currentFrameSlot is reset to 0
8518 e.renderbuffer.texture = d->msaaTex[i];
8519 rhiD->d->releaseQueue.append(e);
8520 }
8521 d->msaaTex[i] = [rhiD->d->dev newTextureWithDescriptor: desc];
8522 }
8523 [desc release];
8524 }
8525
8526 rhiD->registerResource(this);
8527
8528 return true;
8529}
8530
8532{
8535 info.limits.colorComponentValue.maxColorComponentValue = 1;
8536 info.limits.colorComponentValue.maxPotentialColorComponentValue = 1;
8538 info.sdrWhiteLevel = 200; // typical value, but dummy (don't know the real one); won't matter due to being display-referred
8539
8540 if (m_window) {
8541 // Must use m_window, not window, given this may be called before createOrResize().
8542#if defined(Q_OS_MACOS)
8543 NSView *view = reinterpret_cast<NSView *>(m_window->winId());
8544 NSScreen *screen = view.window.screen;
8545 info.limits.colorComponentValue.maxColorComponentValue = screen.maximumExtendedDynamicRangeColorComponentValue;
8546 info.limits.colorComponentValue.maxPotentialColorComponentValue = screen.maximumPotentialExtendedDynamicRangeColorComponentValue;
8547#elif defined(Q_OS_IOS)
8548 UIView *view = reinterpret_cast<UIView *>(m_window->winId());
8549 UIScreen *screen = view.window.windowScene.screen;
8550 info.limits.colorComponentValue.maxColorComponentValue =
8551 view.window.windowScene.screen.currentEDRHeadroom;
8552 info.limits.colorComponentValue.maxPotentialColorComponentValue =
8553 screen.potentialEDRHeadroom;
8554#endif
8555 }
8556
8557 return info;
8558}
8559
8560QT_END_NAMESPACE
const char * constData() const
Definition qrhi_p.h:477
void buildIndirect(QRhiCommandBuffer *cb, QRhiIndirectCommandBuffer *icb, const QRhiIndirectCommandBufferBuildInfo &info) override
void drawIndirectCount(QRhiCommandBuffer *cb, QRhiBuffer *indirectBuffer, quint32 indirectBufferOffset, QRhiBuffer *countBuffer, quint32 countBufferOffset, quint32 maxDrawCount, quint32 stride) override
static QRhiSwapChainProxyData updateSwapChainProxyData(QWindow *window)
QMetalSwapChain * currentSwapChain
bool isDeviceLost() const override
void setBlendConstants(QRhiCommandBuffer *cb, const QColor &c) override
QRhiStats statistics() override
void drawIndexed(QRhiCommandBuffer *cb, quint32 indexCount, quint32 instanceCount, quint32 firstIndex, qint32 vertexOffset, quint32 firstInstance) override
void executeBufferHostWritesForCurrentFrame(QMetalBuffer *bufD)
int ubufAlignment() const override
Definition qrhimetal.mm:851
bool isTextureFormatSupported(QRhiTexture::Format format, QRhiTexture::Flags flags) const override
Definition qrhimetal.mm:882
void endExternal(QRhiCommandBuffer *cb) override
QRhiMetalData * d
void drawIndirect(QRhiCommandBuffer *cb, QRhiBuffer *indirectBuffer, quint32 indirectBufferOffset, quint32 drawCount, quint32 stride) override
QRhiMetal(QRhiMetalInitParams *params, QRhiMetalNativeHandles *importDevice=nullptr)
Definition qrhimetal.mm:566
void beginPass(QRhiCommandBuffer *cb, QRhiRenderTarget *rt, const QColor &colorClearValue, const QRhiDepthStencilClearValue &depthStencilClearValue, QRhiResourceUpdateBatch *resourceUpdates, QRhiCommandBuffer::BeginPassFlags flags) override
qsizetype subresUploadByteSize(const QRhiTextureSubresourceUploadDescription &subresDesc) const
void beginExternal(QRhiCommandBuffer *cb) override
void adjustForMultiViewDraw(quint32 *instanceCount, QRhiCommandBuffer *cb)
void setDefaultScissor(QMetalCommandBuffer *cbD)
void enqueueShaderResourceBindings(QMetalShaderResourceBindings *srbD, QMetalCommandBuffer *cbD, int dynamicOffsetCount, const QRhiCommandBuffer::DynamicOffset *dynamicOffsets, bool offsetOnlyChange, const QShader::NativeResourceBindingMap *nativeResourceBindingMaps[SUPPORTED_STAGES], const QMetalShader *shaders[SUPPORTED_STAGES])
QRhiSwapChain * createSwapChain() override
Definition qrhimetal.mm:841
QRhiGraphicsPipeline * createGraphicsPipeline() override
const char * icbUnavailableReason(QMetalCommandBuffer *cbD) const
bool create(QRhi::Flags flags) override
Definition qrhimetal.mm:642
QRhi::FrameOpResult beginOffscreenFrame(QRhiCommandBuffer **cb, QRhi::BeginFrameFlags flags) override
void dispatchIndirect(QRhiCommandBuffer *cb, QRhiBuffer *indirectBuffer, quint32 indirectBufferOffset) override
void dispatch(QRhiCommandBuffer *cb, int x, int y, int z) override
void resourceUpdate(QRhiCommandBuffer *cb, QRhiResourceUpdateBatch *resourceUpdates) override
QRhiSampler * createSampler(QRhiSampler::Filter magFilter, QRhiSampler::Filter minFilter, QRhiSampler::Filter mipmapMode, QRhiSampler::AddressMode u, QRhiSampler::AddressMode v, QRhiSampler::AddressMode w) override
void setPushConstants(QRhiCommandBuffer *cb, quint32 offset, quint32 size, const void *data) override
static const int SUPPORTED_STAGES
void setGraphicsPipeline(QRhiCommandBuffer *cb, QRhiGraphicsPipeline *ps) override
bool isYUpInNDC() const override
Definition qrhimetal.mm:861
void commitIndirectCommandBuffer(QRhiResourceUpdateBatch *u, QRhiIndirectCommandBuffer *icb) override
int resourceLimit(QRhi::ResourceLimit limit) const override
QRhiShaderResourceBindings * createShaderResourceBindings() override
void executeBufferHostWritesForSlot(QMetalBuffer *bufD, int slot)
QRhi::FrameOpResult beginFrame(QRhiSwapChain *swapChain, QRhi::BeginFrameFlags flags) override
QMatrix4x4 clipSpaceCorrMatrix() const override
Definition qrhimetal.mm:871
void drawIndexedIndirectCount(QRhiCommandBuffer *cb, QRhiBuffer *indirectBuffer, quint32 indirectBufferOffset, QRhiBuffer *countBuffer, quint32 countBufferOffset, quint32 maxDrawCount, quint32 stride) override
const QRhiNativeHandles * nativeHandles() override
void interruptRenderPass(QMetalCommandBuffer *cbD)
void executeDeferredReleases(bool forced=false)
void endPass(QRhiCommandBuffer *cb, QRhiResourceUpdateBatch *resourceUpdates) override
QRhiComputePipeline * createComputePipeline() override
void drawIndexedIndirect(QRhiCommandBuffer *cb, QRhiBuffer *indirectBuffer, quint32 indirectBufferOffset, quint32 drawCount, quint32 stride) override
bool isClipDepthZeroToOne() const override
Definition qrhimetal.mm:866
QRhiTextureRenderTarget * createTextureRenderTarget(const QRhiTextureRenderTargetDescription &desc, QRhiTextureRenderTarget::Flags flags) override
bool isYUpInFramebuffer() const override
Definition qrhimetal.mm:856
bool prepareIcbKernels()
QRhi::FrameOpResult endOffscreenFrame(QRhi::EndFrameFlags flags) override
const QRhiNativeHandles * nativeHandles(QRhiCommandBuffer *cb) override
void enqueueResourceUpdates(QRhiCommandBuffer *cb, QRhiResourceUpdateBatch *resourceUpdates)
void setPipelineCacheData(const QByteArray &data) override
void finalizeDeferredStoreActions(QMetalCommandBuffer *cbD, bool passIsEnding)
QByteArray pipelineCacheData() override
void setShadingRate(QRhiCommandBuffer *cb, const QSize &coarsePixelSize) override
void setScissor(QRhiCommandBuffer *cb, const QRhiScissor &scissor) override
void setVertexInput(QRhiCommandBuffer *cb, int startBinding, int bindingCount, const QRhiCommandBuffer::VertexInput *bindings, QRhiBuffer *indexBuf, quint32 indexOffset, QRhiCommandBuffer::IndexFormat indexFormat) override
void setComputePipeline(QRhiCommandBuffer *cb, QRhiComputePipeline *ps) override
bool importedDevice
void tessellatedDraw(const TessDrawArgs &args)
void debugMarkBegin(QRhiCommandBuffer *cb, const QByteArray &name) override
void debugMarkEnd(QRhiCommandBuffer *cb) override
QRhi::FrameOpResult finish() override
QRhiShadingRateMap * createShadingRateMap() override
bool importedCmdQueue
QRhi::FrameOpResult endFrame(QRhiSwapChain *swapChain, QRhi::EndFrameFlags flags) override
bool makeThreadLocalNativeContextCurrent() override
QRhiTexture * createTexture(QRhiTexture::Format format, const QSize &pixelSize, int depth, int arraySize, int sampleCount, QRhiTexture::Flags flags) override
void setShaderResources(QRhiCommandBuffer *cb, QRhiShaderResourceBindings *srb, int dynamicOffsetCount, const QRhiCommandBuffer::DynamicOffset *dynamicOffsets) override
void releaseCachedResources() override
bool icbDraw(QMetalCommandBuffer *cbD, bool indexed, QMetalBuffer *indirectBufD, quint32 indirectBufferOffset, QMetalBuffer *countBufD, quint32 countBufferOffset, quint32 maxDrawCount, quint32 stride)
void setStencilRef(QRhiCommandBuffer *cb, quint32 refValue) override
void setViewport(QRhiCommandBuffer *cb, const QRhiViewport &viewport) override
void endComputePass(QRhiCommandBuffer *cb, QRhiResourceUpdateBatch *resourceUpdates) override
QRhiDriverInfo driverInfo() const override
double lastCompletedGpuTime(QRhiCommandBuffer *cb) override
QList< int > supportedSampleCounts() const override
void draw(QRhiCommandBuffer *cb, quint32 vertexCount, quint32 instanceCount, quint32 firstVertex, quint32 firstInstance) override
void executeIndirect(QRhiCommandBuffer *cb, QRhiIndirectCommandBuffer *icb, quint32 firstCommand, quint32 commandCount) override
bool prepareIcb(quint32 maxDrawCount)
void finishActiveReadbacks(bool forced=false)
void debugMarkMsg(QRhiCommandBuffer *cb, const QByteArray &msg) override
bool isFeatureSupported(QRhi::Feature feature) const override
Definition qrhimetal.mm:915
void enqueueSubresUpload(QMetalTexture *texD, void *mp, void *blitEncPtr, int layer, int level, const QRhiTextureSubresourceUploadDescription &subresDesc, qsizetype *curOfs)
void destroy() override
Definition qrhimetal.mm:764
QList< QSize > supportedShadingRates(int sampleCount) const override
Definition qrhimetal.mm:835
void beginComputePass(QRhiCommandBuffer *cb, QRhiResourceUpdateBatch *resourceUpdates, QRhiCommandBuffer::BeginPassFlags flags) override
static QRhiResourceUpdateBatchPrivate * get(QRhiResourceUpdateBatch *b)
Definition qrhi_p.h:745
\inmodule QtGuiPrivate \inheaderfile rhi/qrhi.h
Definition qrhi.h:325
\inmodule QtGui
Definition qshader.h:81
Combined button and popup list for selecting options.
#define __has_feature(x)
@ UnBounded
Definition qrhi_p.h:390
@ Bounded
Definition qrhi_p.h:391
#define QRHI_RES_RHI(t)
Definition qrhi_p.h:33
#define QRHI_RES(t, x)
Definition qrhi_p.h:32
Int aligned(Int v, Int byteAlign)
\variable QRhiVulkanQueueSubmitParams::waitSemaphoreCount
static int mtlPushConstantBufferIndex(const QMetalShader &s)
static bool usesTextures(const QShaderDescription &desc)
static MTLStencilOperation toMetalStencilOp(QRhiGraphicsPipeline::StencilOp op)
static void qrhimtl_releaseIcbSlots(QRhiMetal *rhiD, QMetalIndirectCommandBuffer *icbD)
static MTLLanguageVersion toMetalLanguageVersion(const QShaderVersion &version)
static MTLPrimitiveTopologyClass toMetalPrimitiveTopologyClass(QRhiGraphicsPipeline::Topology t)
static CAMetalLayer * layerForWindow(QWindow *window)
static void addVertexAttribute(const T &variable, int binding, QRhiMetal *rhiD, int &index, quint32 &offset, MTLVertexAttributeDescriptorArray *attributes, quint64 &indices, quint32 &vertexAlignment)
static void qrhimtl_releaseRenderBuffer(const QRhiMetalData::DeferredReleaseEntry &e)
static void qrhimtl_encodeIcbFromCpu(QMetalIndirectCommandBuffer *icbD, QMetalIndirectCommandBufferData::Slot &slot, MTLPrimitiveType primitiveType, id< MTLBuffer > indexBufMtl, quint32 indexOffset, QRhiCommandBuffer::IndexFormat indexFormat)
static bool hasArgumentBufferVariant(const QShader &shader)
static bool matches(const QList< QShaderDescription::BlockVariable > &a, const QList< QShaderDescription::BlockVariable > &b)
Q_DECLARE_TYPEINFO(QRhiMetalData::TextureReadback, Q_RELOCATABLE_TYPE)
static MTLBlendOperation toMetalBlendOp(QRhiGraphicsPipeline::BlendOp op)
static QMetalRenderTargetData * currentRenderTargetData(QMetalCommandBuffer *cbD)
static MTLBlendFactor toMetalBlendFactor(QRhiGraphicsPipeline::BlendFactor f)
static MTLWinding toMetalTessellationWindingOrder(QShaderDescription::TessellationWindingOrder w)
static MTLPrimitiveType toMetalPrimitiveType(QRhiGraphicsPipeline::Topology t)
static MTLCompareFunction toMetalCompareOp(QRhiGraphicsPipeline::CompareOp op)
static MTLVertexFormat toMetalAttributeFormat(QRhiVertexInputAttribute::Format format)
static constexpr quint32 ICB_DRAW_COUNT_THRESHOLD
BindingType
static MTLTriangleFillMode toMetalTriangleFillMode(QRhiGraphicsPipeline::PolygonMode mode)
static quint32 mtlPushConstantBlockSize(const QMetalShader &s)
static MTLSamplerMinMagFilter toMetalFilter(QRhiSampler::Filter f)
static MTLCullMode toMetalCullMode(QRhiGraphicsPipeline::CullMode c)
static void qrhimtl_releaseBuffer(const QRhiMetalData::DeferredReleaseEntry &e)
static void endTempComputeEncoding(QRhiMetal *rhiD, QMetalCommandBuffer *cbD, id< MTLComputeCommandEncoder > computeEncoder)
static void takeIndex(quint32 index, quint64 &indices)
static int mapBinding(int binding, int stageIndex, const QShader::NativeResourceBindingMap *nativeResourceBindingMaps[], BindingType type)
static id< MTLComputeCommandEncoder > tempComputeEncoder(QRhiMetal *rhiD, QMetalCommandBuffer *cbD, id< MTLComputeCommandEncoder > maybeComputeEncoder)
static void rebindShaderResources(QMetalCommandBuffer *cbD, int resourceStage, int encoderStage, const QMetalShaderResourceBindingsData *customBindingState=nullptr)
static void qrhimtl_releaseSampler(const QRhiMetalData::DeferredReleaseEntry &e)
static MTLPixelFormat toMetalTextureFormat(QRhiTexture::Format format, QRhiTexture::Flags flags, const QRhiMetal *d)
static QRhiShaderResourceBinding::StageFlag toRhiSrbStage(int stage)
static void setupArgumentBufferEncoder(QMetalShader *shader)
static void addUnusedVertexAttribute(const T &variable, QRhiMetal *rhiD, quint32 &offset, quint32 &vertexAlignment)
static uint toMetalColorWriteMask(QRhiGraphicsPipeline::ColorMask c)
static MTLStoreAction interruptionStoreAction(MTLStoreAction finalAction, bool passIsEnding)
#define QRHI_METAL_COMMAND_BUFFERS_WITH_UNRETAINED_REFERENCES
Definition qrhimetal.mm:61
static void qrhimtl_replayIcbOnCpu(QMetalCommandBuffer *cbD, QMetalIndirectCommandBuffer *icbD, quint32 firstCommand, quint32 count, int currentFrameSlot)
static bool qrhimtl_ensureIcbSlots(QRhiMetal *rhiD, QMetalIndirectCommandBuffer *icbD, QMetalIndirectCommandBufferData::Fill fill)
static MTLSamplerMipFilter toMetalMipmapMode(QRhiSampler::Filter f)
static MTLTessellationPartitionMode toMetalTessellationPartitionMode(QShaderDescription::TessellationPartitioning p)
static MTLCompareFunction toMetalTextureCompareFunction(QRhiSampler::CompareOp op)
static int aligned(quint32 offset, quint32 alignment)
static void declareStageArgumentBufferResources(QMetalCommandBuffer *cbD, int encoderStage, const QMetalShaderResourceBindingsData::Stage &res)
static bool canStoreAttachment(id< MTLTexture > tex)
Q_DECLARE_TYPEINFO(QRhiMetalData::DeferredReleaseEntry, Q_RELOCATABLE_TYPE)
static MTLResourceUsage storageImageUsage(QRhiShaderResourceBinding::Type type)
static bool qrhimtl_icbSlotMatches(const QMetalIndirectCommandBuffer *icbD, const QMetalIndirectCommandBufferData::Slot &slot, MTLPrimitiveType primitiveType, id< MTLBuffer > indexBufMtl, quint32 indexOffset, QRhiCommandBuffer::IndexFormat indexFormat)
static MTLSamplerAddressMode toMetalAddressMode(QRhiSampler::AddressMode m)
static void bindStageBuffers(QMetalCommandBuffer *cbD, int stage, const QRhiBatchedBindings< id< MTLBuffer > >::Batch &bufferBatch, const QRhiBatchedBindings< NSUInteger >::Batch &offsetBatch)
static void qrhimtl_releaseTexture(const QRhiMetalData::DeferredReleaseEntry &e)
static void encodeIcbWithCompute(QRhiMetalData *d, id< MTLComputeCommandEncoder > computeEncoder, id< MTLIndirectCommandBuffer > targetIcb, id< MTLBuffer > targetArgBuffer, id< MTLBuffer > targetRangeBuffer, bool indexed, QRhiCommandBuffer::IndexFormat indexFormat, MTLPrimitiveType primitiveType, id< MTLBuffer > indirectBufMtl, quint32 indirectBufferOffset, id< MTLBuffer > indexBufMtl, quint32 indexBufferOffset, id< MTLBuffer > countBufMtl, quint32 countBufferOffset, quint32 maxDrawCount, quint32 stride)
static bool indexTaken(quint32 index, quint64 indices)
static void bindStageTextures(QMetalCommandBuffer *cbD, int stage, const QRhiBatchedBindings< id< MTLTexture > >::Batch &textureBatch)
#define QRHI_METAL_DISABLE_BINARY_ARCHIVE
Definition qrhimetal.mm:56
static void bindStageSamplers(QMetalCommandBuffer *cbD, int encoderStage, const QRhiBatchedBindings< id< MTLSamplerState > >::Batch &samplerBatch)
static int nextAttributeIndex(quint64 indices)
static QT_BEGIN_NAMESPACE const int QMTL_FRAMES_IN_FLIGHT
Definition qrhimetal_p.h:24
void f(int c)
[26]
QVarLengthArray< BufferUpdate, 16 > pendingUpdates[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:354
id< MTLBuffer > buf[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:349
char * beginFullDynamicBufferUpdateForCurrentFrame() override
QMetalBufferData * d
Definition qrhimetal_p.h:39
QMetalBuffer(QRhiImplementation *rhi, Type type, UsageFlags usage, quint32 size)
int lastActiveFrameSlot
Definition qrhimetal_p.h:41
QRhiBuffer::NativeBuffer nativeBuffer() override
void endFullDynamicBufferUpdateForCurrentFrame() override
To be called when the entire contents of the buffer data has been updated in the memory block returne...
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
bool create() override
Creates the corresponding native graphics resources.
MTLRenderPassDescriptor * currentPassRpDesc
Definition qrhimetal.mm:430
id< MTLDepthStencilState > currentDepthStencilState
Definition qrhimetal.mm:434
QMetalShaderResourceBindingsData currentShaderResourceBindingState
Definition qrhimetal.mm:435
MTLStoreAction deferredStencilStoreAction
Definition qrhimetal.mm:443
QVarLengthArray< std::pair< uint, MTLStoreAction >, 4 > deferredColorStoreActions
Definition qrhimetal.mm:441
MTLStoreAction deferredDepthStoreAction
Definition qrhimetal.mm:442
id< MTLComputeCommandEncoder > tessellationComputeEncoder
Definition qrhimetal.mm:429
QRhiBatchedBindings< id< MTLBuffer > > currentVertexInputsBuffers
Definition qrhimetal.mm:432
id< MTLRenderCommandEncoder > currentRenderPassEncoder
Definition qrhimetal.mm:427
id< MTLCommandBuffer > cb
Definition qrhimetal.mm:425
QRhiBatchedBindings< NSUInteger > currentVertexInputOffsets
Definition qrhimetal.mm:433
QVarLengthArray< QByteArray, 4 > openDebugGroups
Definition qrhimetal.mm:436
id< MTLComputeCommandEncoder > currentComputePassEncoder
Definition qrhimetal.mm:428
QMetalBuffer * currentIndexBuffer
const QRhiNativeHandles * nativeHandles()
QMetalShaderResourceBindings * currentComputeSrb
QMetalComputePipeline * currentComputePipeline
QMetalShaderResourceBindings * currentGraphicsSrb
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
QMetalCommandBuffer(QRhiImplementation *rhi)
void resetPerPassCachedState()
QMetalCommandBufferData * d
QMetalGraphicsPipeline * currentGraphicsPipeline
void resetState(double lastGpuTime=0)
id< MTLComputePipelineState > ps
Definition qrhimetal.mm:546
QMetalBuffer * bufferSizeBuffer
Definition qrhimetal.mm:551
QMetalComputePipeline(QRhiImplementation *rhi)
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
QMetalComputePipelineData * d
bool create() override
QVector< QMetalBuffer * > deviceLocalWorkBuffers
Definition qrhimetal.mm:500
QMetalBuffer * acquireWorkBuffer(QRhiMetal *rhiD, quint32 size, WorkBufType type=WorkBufType::DeviceLocal)
QVector< QMetalBuffer * > hostVisibleWorkBuffers
Definition qrhimetal.mm:501
quint32 tescCompOutputBufferSize(quint32 patchCount) const
Definition qrhimetal.mm:519
std::array< id< MTLComputePipelineState >, 3 > vertexComputeState
Definition qrhimetal.mm:510
quint32 tescCompPatchOutputBufferSize(quint32 patchCount) const
Definition qrhimetal.mm:523
static int vsCompVariantToIndex(QShader::Variant vertexCompVariant)
id< MTLComputePipelineState > tescCompPipeline(QRhiMetal *rhiD)
id< MTLRenderPipelineState > teseFragRenderPipeline(QRhiMetal *rhiD, QMetalGraphicsPipeline *pipeline)
QMetalGraphicsPipelineData * q
Definition qrhimetal.mm:504
id< MTLComputePipelineState > vsCompPipeline(QRhiMetal *rhiD, QShader::Variant vertexCompVariant)
quint32 patchCountForDrawCall(quint32 vertexOrIndexCount, quint32 instanceCount) const
Definition qrhimetal.mm:528
quint32 vsCompOutputBufferSize(quint32 vertexOrIndexCount, quint32 instanceCount) const
Definition qrhimetal.mm:514
id< MTLComputePipelineState > tessControlComputeState
Definition qrhimetal.mm:511
QMetalGraphicsPipeline * q
Definition qrhimetal.mm:481
MTLDepthClipMode depthClipMode
Definition qrhimetal.mm:489
MTLPrimitiveType primitiveType
Definition qrhimetal.mm:485
id< MTLRenderPipelineState > ps
Definition qrhimetal.mm:482
QMetalBuffer * bufferSizeBuffer
Definition qrhimetal.mm:541
void setupVertexInputDescriptor(MTLVertexDescriptor *desc)
void setupStageInputDescriptor(MTLStageInputOutputDescriptor *desc)
id< MTLDepthStencilState > ds
Definition qrhimetal.mm:483
MTLTriangleFillMode triangleFillMode
Definition qrhimetal.mm:488
QMetalGraphicsPipelineData * d
bool createVertexFragmentPipeline()
QMetalGraphicsPipeline(QRhiImplementation *rhi)
void setupAttachmentsInMetalRenderPassDescriptor(void *metalRpDesc, QMetalRenderPassDescriptor *rpD)
void makeActiveForCurrentRenderPassEncoder(QMetalCommandBuffer *cbD)
bool create() override
Creates the corresponding native graphics resources.
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
void setupMetalDepthStencilDescriptor(void *metalDsDesc)
bool createTessellationPipelines(const QShader &tessVert, const QShader &tesc, const QShader &tese, const QShader &tessFrag)
QRhiCommandBuffer::IndexFormat indexFormat
id< MTLIndirectCommandBuffer > icb
QRhiIndirectCommandBufferBuildInfo buildInfo
Slot frameSlots[QMTL_FRAMES_IN_FLIGHT]
QMetalIndirectCommandBuffer(QRhiImplementation *rhi, Type type, quint32 maxCommandCount)
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
QMetalIndirectCommandBufferData * d
const QRhiIndexedIndirectDrawCommand * indexedDrawCommands() const
bool create() override
Creates the corresponding native objects.
id< MTLTexture > tex
Definition qrhimetal.mm:360
MTLPixelFormat format
Definition qrhimetal.mm:359
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
QMetalRenderBufferData * d
Definition qrhimetal_p.h:67
QRhiTexture::Format backingFormat() const override
bool create() override
Creates the corresponding native graphics resources.
QMetalRenderBuffer(QRhiImplementation *rhi, Type type, const QSize &pixelSize, int sampleCount, QRhiRenderBuffer::Flags flags, QRhiTexture::Format backingFormatHint)
QMetalRenderPassDescriptor(QRhiImplementation *rhi)
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
QVector< quint32 > serializedFormat() const override
bool isCompatible(const QRhiRenderPassDescriptor *other) const override
int colorFormat[MAX_COLOR_ATTACHMENTS]
static const int MAX_COLOR_ATTACHMENTS
QRhiRenderPassDescriptor * newCompatibleRenderPassDescriptor() const override
ColorAtt colorAtt[QMetalRenderPassDescriptor::MAX_COLOR_ATTACHMENTS]
Definition qrhimetal.mm:467
id< MTLTexture > dsResolveTex
Definition qrhimetal.mm:469
QRhiRenderTargetAttachmentTracker::ResIdList currentResIdList
Definition qrhimetal.mm:476
id< MTLTexture > dsTex
Definition qrhimetal.mm:468
id< MTLSamplerState > samplerState
Definition qrhimetal.mm:387
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
QMetalSampler(QRhiImplementation *rhi, Filter magFilter, Filter minFilter, Filter mipmapMode, AddressMode u, AddressMode v, AddressMode w)
QMetalSamplerData * d
int lastActiveFrameSlot
bool create() override
QVarLengthArray< Buffer, 8 > buffers
Definition qrhimetal.mm:411
QVarLengthArray< Sampler, 8 > samplers
Definition qrhimetal.mm:413
QRhiBatchedBindings< NSUInteger > bufferOffsetBatches
Definition qrhimetal.mm:415
QVarLengthArray< Texture, 8 > textures
Definition qrhimetal.mm:412
QRhiBatchedBindings< id< MTLSamplerState > > samplerBatches
Definition qrhimetal.mm:417
QRhiBatchedBindings< id< MTLTexture > > textureBatches
Definition qrhimetal.mm:416
QRhiBatchedBindings< id< MTLBuffer > > bufferBatches
Definition qrhimetal.mm:414
bool create() override
Creates the corresponding resource binding set.
QMetalComputePipeline * lastUsedComputePipeline
QMetalShaderResourceBindings(QRhiImplementation *rhi)
QMetalGraphicsPipeline * lastUsedGraphicsPipeline
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
void updateResources(UpdateFlags flags) override
\variable QRhiMetalCommandBufferNativeHandles::commandBuffer
Definition qrhimetal.mm:156
void destroy()
Definition qrhimetal.mm:167
id< MTLLibrary > lib
Definition qrhimetal.mm:157
uint outputVertexCount
Definition qrhimetal.mm:160
std::array< uint, 3 > localSize
Definition qrhimetal.mm:159
QShaderDescription desc
Definition qrhimetal.mm:161
id< MTLFunction > func
Definition qrhimetal.mm:158
int argumentBufferIndex
Definition qrhimetal.mm:165
id< MTLArgumentEncoder > argumentEncoder
Definition qrhimetal.mm:164
id< MTLRasterizationRateMap > rateMap
Definition qrhimetal.mm:392
QMetalShadingRateMapData * d
QMetalShadingRateMap(QRhiImplementation *rhi)
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
QSize logicalSize() const override
bool createFrom(NativeShadingRateMap src) override
Sets up the shading rate map to use a native 3D API shading rate object src.
id< CAMetalDrawable > curDrawable
Definition qrhimetal.mm:557
dispatch_semaphore_t sem[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:558
MTLPixelFormat colorFormat
Definition qrhimetal.mm:563
MTLRenderPassDescriptor * rp
Definition qrhimetal.mm:560
CAMetalLayer * layer
Definition qrhimetal.mm:556
double lastGpuTime[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:559
id< MTLTexture > msaaTex[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:561
QRhiTexture::Format rhiColorFormat
Definition qrhimetal.mm:562
QMetalRenderTargetData * d
QMetalSwapChainRenderTarget(QRhiImplementation *rhi, QRhiSwapChain *swapchain)
QSize pixelSize() const override
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
float devicePixelRatio() const override
int sampleCount() const override
void waitUntilCompleted(int slot)
bool createOrResize() override
Creates the swapchain if not already done and resizes the swapchain buffers to match the current size...
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
QRhiCommandBuffer * currentFrameCommandBuffer() override
QMetalSwapChain(QRhiImplementation *rhi)
virtual QRhiSwapChainHdrInfo hdrInfo() override
\variable QRhiSwapChainHdrInfo::limitsType
QMetalRenderBuffer * ds
QMetalSwapChainRenderTarget rtWrapper
QRhiRenderPassDescriptor * newCompatibleRenderPassDescriptor() override
QMetalSwapChainData * d
bool isFormatSupported(Format f) override
QSize surfacePixelSize() override
QRhiRenderTarget * currentFrameRenderTarget() override
MTLPixelFormat viewFormat
Definition qrhimetal.mm:369
id< MTLTexture > tex
Definition qrhimetal.mm:371
MTLPixelFormat viewFormatForSampling
Definition qrhimetal.mm:370
id< MTLTexture > viewForLevel(int level)
QMetalTexture * q
Definition qrhimetal.mm:367
id< MTLTexture > perLevelViews[QRhi::MAX_MIP_LEVELS]
Definition qrhimetal.mm:376
id< MTLBuffer > stagingBuf[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:374
id< MTLTexture > writeView
Definition qrhimetal.mm:373
QMetalTextureData(QMetalTexture *t)
Definition qrhimetal.mm:365
id< MTLTexture > samplingView
Definition qrhimetal.mm:372
id< MTLTexture > textureForWrite() const
Definition qrhimetal.mm:379
MTLPixelFormat format
Definition qrhimetal.mm:368
id< MTLTexture > textureForSampling() const
Definition qrhimetal.mm:378
float devicePixelRatio() const override
QMetalRenderTargetData * d
QMetalTextureRenderTarget(QRhiImplementation *rhi, const QRhiTextureRenderTargetDescription &desc, Flags flags)
bool create() override
Creates the corresponding native graphics resources.
QRhiRenderPassDescriptor * newCompatibleRenderPassDescriptor() override
int sampleCount() const override
QSize pixelSize() const override
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
QMetalTexture(QRhiImplementation *rhi, Format format, const QSize &pixelSize, int depth, int arraySize, int sampleCount, Flags flags)
bool prepareCreate(QSize *adjustedSize=nullptr)
NativeTexture nativeTexture() override
QMetalTextureData * d
Definition qrhimetal_p.h:88
bool create() override
Creates the corresponding native graphics resources.
int lastActiveFrameSlot
Definition qrhimetal_p.h:92
void destroy() override
Releases (or requests deferred releasing of) the underlying native graphics resources.
bool createFrom(NativeTexture src) override
Similar to create(), except that no new native textures are created.
\variable QRhiIndirectDrawCommand::vertexCount
Definition qrhi.h:1740
QRhiReadbackResult * result
Definition qrhimetal.mm:279
id< MTLComputePipelineState > pipelineState
Definition qrhimetal.mm:245
id< MTLDepthStencilState > depthStencilState
Definition qrhimetal.mm:240
std::array< id< MTLComputePipelineState >, 3 > tessVertexComputeState
Definition qrhimetal.mm:241
id< MTLRasterizationRateMap > rateMap
Definition qrhimetal.mm:248
id< MTLSamplerState > samplerState
Definition qrhimetal.mm:233
id< MTLBuffer > stagingBuffers[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:227
id< MTLComputePipelineState > tessTessControlComputeState
Definition qrhimetal.mm:242
id< MTLIndirectCommandBuffer > icb
Definition qrhimetal.mm:251
id< MTLRenderPipelineState > pipelineState
Definition qrhimetal.mm:239
id< MTLBuffer > buffers[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:220
id< MTLTexture > views[QRhi::MAX_MIP_LEVELS]
Definition qrhimetal.mm:228
QMetalCommandBuffer cbWrapper
Definition qrhimetal.mm:262
OffscreenFrame(QRhiImplementation *rhi)
Definition qrhimetal.mm:259
QRhiReadbackDescription desc
Definition qrhimetal.mm:267
QRhiReadbackResult * result
Definition qrhimetal.mm:268
QRhiTexture::Format format
Definition qrhimetal.mm:272
void trySeedingRenderPipelineFromBinaryArchive(MTLRenderPipelineDescriptor *rpDesc)
id< MTLComputePipelineState > icbEncodePipelineU32
Definition qrhimetal.mm:293
static constexpr quint32 STAGING_AREA_MIN
Definition qrhimetal.mm:331
QRhiMetalData(QRhiMetal *rhi)
Definition qrhimetal.mm:181
QVarLengthArray< BufferReadback, 2 > activeBufferReadbacks
Definition qrhimetal.mm:284
QHash< ShaderCacheKey, QMetalShader > shaderCache
Definition qrhimetal.mm:308
bool setupBinaryArchive(NSURL *sourceFileUrl=nil)
Definition qrhimetal.mm:622
bool icbSetupFailed
Definition qrhimetal.mm:303
id< MTLFunction > icbEncodeFunctionU16
Definition qrhimetal.mm:297
void addRenderPipelineToBinaryArchive(MTLRenderPipelineDescriptor *rpDesc)
id< MTLFunction > icbEncodeFunctionU32
Definition qrhimetal.mm:296
MTLCaptureManager * captureMgr
Definition qrhimetal.mm:286
id< MTLBuffer > icbArgumentBuffer
Definition qrhimetal.mm:298
static constexpr quint32 LARGE_STAGING_ALLOC
Definition qrhimetal.mm:330
NSUInteger icbCapacity
Definition qrhimetal.mm:291
void trySeedingComputePipelineFromBinaryArchive(MTLComputePipelineDescriptor *cpDesc)
id< MTLIndirectCommandBuffer > icb
Definition qrhimetal.mm:290
QVector< DeferredReleaseEntry > releaseQueue
Definition qrhimetal.mm:256
id< MTLBuffer > allocArgumentBuffer(quint32 size, quint32 alignment, int frameSlot, quint32 *offset)
static constexpr quint32 STAGING_AREA_MAX
Definition qrhimetal.mm:332
id< MTLLibrary > createMetalLib(const QShader &shader, QShader::Variant shaderVariant, bool preferArgumentBuffers, QString *error, QByteArray *entryPoint, QShaderKey *activeKey)
id< MTLFunction > createMSLShaderFunction(id< MTLLibrary > lib, const QByteArray &entryPoint)
id< MTLBuffer > allocBufferStaging(quint32 size, int frameSlot, quint32 *offset)
id< MTLCaptureScope > captureScope
Definition qrhimetal.mm:287
void resetAndResizeBufferStagingArea(int frameSlot)
MTLRenderPassDescriptor * createDefaultRenderPass(bool hasDepthStencil, const QColor &colorClearValue, const QRhiDepthStencilClearValue &depthStencilClearValue, int colorAttCount, QRhiShadingRateMap *shadingRateMap)
id< MTLBuffer > newOneShotStagingBuffer(quint32 size, int frameSlot)
QRhiMetal * q
Definition qrhimetal.mm:183
id< MTLComputePipelineState > icbEncodePipeline
Definition qrhimetal.mm:292
static constexpr int STAGING_AREA_HIGH_DEMAND_FRAMES
Definition qrhimetal.mm:334
id< MTLFunction > icbEncodeFunction
Definition qrhimetal.mm:295
id< MTLComputePipelineState > icbEncodePipelineU16
Definition qrhimetal.mm:294
static const int TEXBUF_ALIGN
Definition qrhimetal.mm:305
id< MTLBuffer > icbRangeBuffer
Definition qrhimetal.mm:300
id< MTLBuffer > allocFromStagingArea(StagingArea *area, quint32 size, quint32 alignment, int frameSlot, quint32 minBlockSize, quint32 *offset)
StagingArea bufStagingPool[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:319
id< MTLBinaryArchive > binArch
Definition qrhimetal.mm:186
id< MTLCommandBuffer > newCommandBuffer()
Definition qrhimetal.mm:610
QVarLengthArray< TextureReadback, 2 > activeTextureReadbacks
Definition qrhimetal.mm:274
StagingArea argBufPool[QMTL_FRAMES_IN_FLIGHT]
Definition qrhimetal.mm:318
id< MTLDevice > dev
Definition qrhimetal.mm:184
quint64 globalFrameId
Definition qrhimetal.mm:338
static constexpr int STAGING_AREA_LOW_DEMAND_FRAMES
Definition qrhimetal.mm:333
void addComputePipelineToBinaryArchive(MTLComputePipelineDescriptor *cpDesc)
id< MTLCommandQueue > cmdQueue
Definition qrhimetal.mm:185
id< MTLBuffer > icbNoCountBuffer
Definition qrhimetal.mm:302
QMetalCommandBuffer * cbD
\inmodule QtGuiPrivate \inheaderfile rhi/qrhi.h
Definition qrhi.h:1988
\inmodule QtGuiPrivate \inheaderfile rhi/qrhi.h
Definition qrhi.h:1588
LimitsType limitsType
Definition qrhi.h:1599
float maxPotentialColorComponentValue
Definition qrhi.h:1607
LuminanceBehavior luminanceBehavior
Definition qrhi.h:1610
float maxColorComponentValue
Definition qrhi.h:1606
\inmodule QtGuiPrivate \inheaderfile rhi/qrhi.h
Definition qrhi.h:1621